Qwen3-14B-q4f16_1-MLC / ndarray-cache.json
riczhou's picture
Upload folder using huggingface_hub
06c924a verified
{
"metadata": {
"ParamSize": 485,
"ParamBytes": 8307783680.0,
"BitsPerParam": 2.693215451538858
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 388956160,
"records": [
{
"name": "lm_head.q_weight",
"shape": [
151936,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 388956160,
"byteOffset": 0
}
],
"md5sum": "8f54523f343dbde2ab3e2cd6a94ea9b6"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 48619520,
"records": [
{
"name": "lm_head.q_scale",
"shape": [
151936,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 48619520,
"byteOffset": 0
}
],
"md5sum": "dbdb1bc0257f352bf5878e6578233ead"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.39.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "3181820f697fbf6c70e21dfeda8c3acc"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.39.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "d14cc5b4f0af13d23bc77e22f15fc643"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 388956160,
"records": [
{
"name": "model.embed_tokens.q_weight",
"shape": [
151936,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 388956160,
"byteOffset": 0
}
],
"md5sum": "705014a58e468ed1f2d95311103b9a07"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 48619520,
"records": [
{
"name": "model.embed_tokens.q_scale",
"shape": [
151936,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 48619520,
"byteOffset": 0
}
],
"md5sum": "7423a89f000627c05b0280d69df04dfd"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.0.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "f0d70a493ad44e60e145f465f361a8f2"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.0.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "5c635568fa9f2b6eb6994dec650ae625"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.0.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "7b07072c5d6b64271031224e81193433"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 33474816,
"records": [
{
"name": "model.layers.39.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 0
},
{
"name": "model.layers.39.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 10240
},
{
"name": "model.layers.39.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 5580800
},
{
"name": "model.layers.39.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16721920
},
{
"name": "model.norm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16732160
},
{
"name": "model.layers.0.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16742400
},
{
"name": "model.layers.0.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 16752640
},
{
"name": "model.layers.0.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 22323200
},
{
"name": "model.layers.0.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 33464320
},
{
"name": "model.layers.0.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 33474560
}
],
"md5sum": "cf6e11fa35b25ff0ef847aef3b4f93db"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.1.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "4591d09d66abbab9caae95a6b968a1a5"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.1.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "71407383a8adbd4a136f176e3e9b39ca"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.0.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.0.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.0.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.0.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.1.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.1.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "a855b7dc7906291c87341cd60d03e223"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.1.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.1.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.1.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.1.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.1.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "b4b477e28b395b254857138966cc98ab"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.2.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "26e3b9e31c2878434d0a44ca60f18c8a"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.2.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "ddcf2cfd34c075b9b2ee55ace68746b3"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.2.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "85ac1e9c797356a3d04b49f2983c9d20"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.1.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.1.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.1.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.2.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.2.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.2.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.2.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.2.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "8f7778707f86358b87b42589ef2e2bff"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.3.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "b21df5a3775a90b56a6f14ed2d1d5864"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.3.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "ca8a8c0f4ac84f8ba4880707d996f07e"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 30474752,
"records": [
{
"name": "model.layers.2.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.2.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.2.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.2.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.3.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 17039616
},
{
"name": "model.layers.3.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 28180736
},
{
"name": "model.layers.3.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28180992
}
],
"md5sum": "eadb2747a8f8f110661a5bb2b67d260e"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.10.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "68d8476be9e5b52cab43e2bd5b6d8a76"
},
{
"dataPath": "params_shard_22.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.10.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "1ed8d316c73d9190f696a09db599ee33"
},
{
"dataPath": "params_shard_23.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.10.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "5fb438e3e65bffaeb506b8cbcaad5439"
},
{
"dataPath": "params_shard_24.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.3.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.3.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.3.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.10.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.10.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.10.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.10.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.10.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "e26755c7e589ad026e122c9f103ce351"
},
{
"dataPath": "params_shard_25.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.11.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "55bef733960efe6938c8a80cd7096fcf"
},
{
"dataPath": "params_shard_26.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.11.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "2507eb55cae231196a0eddba24fd9023"
},
{
"dataPath": "params_shard_27.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.10.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.10.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.10.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.10.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.11.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.11.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "1bc89eda2e1b5c2e3185c9a6c808fe78"
},
{
"dataPath": "params_shard_28.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.11.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.11.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.11.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.11.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.11.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "34cd667148ee1d67adc249f1518f6554"
},
{
"dataPath": "params_shard_29.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.12.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "cf1f32d3366b3f778c88ffce2c2107fb"
},
{
"dataPath": "params_shard_30.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.12.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "3fb79dde34ef10ab6156512333a75dcd"
},
{
"dataPath": "params_shard_31.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.12.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "e0e7db98c4669e8f897f27055961fa89"
},
{
"dataPath": "params_shard_32.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.11.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.11.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.11.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.12.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.12.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.12.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.12.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.12.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "5492c126b70e1509729146dbb95348de"
},
{
"dataPath": "params_shard_33.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.13.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "c11f0963b671c7bd6bc8b03bcc3f55e8"
},
{
"dataPath": "params_shard_34.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.13.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "7beea37257486fddb998be49a50045ba"
},
{
"dataPath": "params_shard_35.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.12.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.12.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.12.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.12.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.13.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.13.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "b74d7dd0677939317c36a213f84e1dce"
},
{
"dataPath": "params_shard_36.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.13.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.13.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.13.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.13.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.13.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "85afebd6b4bc0048d5d14516fe022820"
},
{
"dataPath": "params_shard_37.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.14.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "3c6a70db5651bbf964635ed5ca6fa30a"
},
{
"dataPath": "params_shard_38.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.14.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "936a537eec2a2318079c1a31c7966293"
},
{
"dataPath": "params_shard_39.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.14.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "a40cb2364287cb3d80f3e7249097dd74"
},
{
"dataPath": "params_shard_40.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.13.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.13.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.13.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.14.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.14.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.14.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.14.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.14.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "d54d78c41cd5ebe42f23109cc472e930"
},
{
"dataPath": "params_shard_41.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.15.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "141d01a7cb415c9f5357b8766ce577b0"
},
{
"dataPath": "params_shard_42.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.15.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "b0880ad304601ab84a2a32e6b3ff2973"
},
{
"dataPath": "params_shard_43.bin",
"format": "raw-shard",
"nbytes": 30474752,
"records": [
{
"name": "model.layers.14.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.14.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.14.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.14.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.15.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 17039616
},
{
"name": "model.layers.15.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 28180736
},
{
"name": "model.layers.15.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28180992
}
],
"md5sum": "dbd64df65d52c372d5aa11c7959c6289"
},
{
"dataPath": "params_shard_44.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.9.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "733b922b97a5f7902688cf61d6fc7c41"
},
{
"dataPath": "params_shard_45.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.9.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "a3406391d370df23cb4d4f1d518927e4"
},
{
"dataPath": "params_shard_46.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.15.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "707d711f205f55656d7e386a1a5caeff"
},
{
"dataPath": "params_shard_47.bin",
"format": "raw-shard",
"nbytes": 31488256,
"records": [
{
"name": "model.layers.15.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.15.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.15.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.9.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.9.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.9.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.9.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.15.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31478016
}
],
"md5sum": "969d0bc6e39c143857f9ead53114e1b3"
},
{
"dataPath": "params_shard_48.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.16.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "c9fd7bac400480553c87fa63f8ad440d"
},
{
"dataPath": "params_shard_49.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.16.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "1dbdbea0d691e9c7fea738ce945faee0"
},
{
"dataPath": "params_shard_50.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.16.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "3c7baf9463b78df6f1c7aec01695c6d5"
},
{
"dataPath": "params_shard_51.bin",
"format": "raw-shard",
"nbytes": 24606976,
"records": [
{
"name": "model.layers.15.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 0
},
{
"name": "model.layers.15.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 5570560
},
{
"name": "model.layers.16.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 5580800
},
{
"name": "model.layers.16.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 5591040
},
{
"name": "model.layers.16.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 11161600
},
{
"name": "model.layers.16.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 22302720
},
{
"name": "model.layers.16.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 22312960
},
{
"name": "model.layers.16.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 22313216
}
],
"md5sum": "eaeb62905b13e09c41a1b4662aaa0e22"
},
{
"dataPath": "params_shard_52.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.17.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "37a71dd59cd5a2bd26573224815f28ce"
},
{
"dataPath": "params_shard_53.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.17.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "d00b830b548277ba4a386b146e8b82cc"
},
{
"dataPath": "params_shard_54.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.17.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "eb1f01bd335404776ca9b76d41c485bd"
},
{
"dataPath": "params_shard_55.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.16.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.16.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.16.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.17.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.17.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.17.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.17.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.17.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "80ff329d444a1fffd8150fb5567d923b"
},
{
"dataPath": "params_shard_56.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.18.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "50b23e5bf4197cbb366f318e2152d250"
},
{
"dataPath": "params_shard_57.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.18.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "fc7f12722caac72f635aa20815d6ea60"
},
{
"dataPath": "params_shard_58.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.17.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.17.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.17.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.17.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.18.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.18.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "659f71076b16074a9a0f6d6eae69b756"
},
{
"dataPath": "params_shard_59.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.18.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.18.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.18.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.18.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.18.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "15e952140f6fe5e6137e421875b9329d"
},
{
"dataPath": "params_shard_60.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.19.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "f4c63a4f9704aaca22e59b2dfe56e983"
},
{
"dataPath": "params_shard_61.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.19.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "d954efaa04b63205c739cb4c1f3a0fce"
},
{
"dataPath": "params_shard_62.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.19.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "8b82267663934e0ff3e82c9b83b32f7e"
},
{
"dataPath": "params_shard_63.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.18.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.18.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.18.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.19.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.19.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.19.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.19.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.19.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "36007f964b97ef4e8fc4c83970140386"
},
{
"dataPath": "params_shard_64.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.20.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "5c036fe4bd3e201cd6c0f7cc0270ae25"
},
{
"dataPath": "params_shard_65.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.20.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "bb29858402e453e9d1f418f0246994cd"
},
{
"dataPath": "params_shard_66.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.19.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.19.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.19.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.19.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.20.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.20.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "c3934b46a0906d8dbe4504d9787f8ed0"
},
{
"dataPath": "params_shard_67.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.20.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.20.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.20.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.20.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.20.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "793c42b434059165b8ec641dd6a3db36"
},
{
"dataPath": "params_shard_68.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.21.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "011bb47cee1e0dd904051b7897a383c6"
},
{
"dataPath": "params_shard_69.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.21.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "b2d294d48cef38c28e7f5ede1db0231f"
},
{
"dataPath": "params_shard_70.bin",
"format": "raw-shard",
"nbytes": 28180992,
"records": [
{
"name": "model.layers.20.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.20.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.20.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.21.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 14745856
},
{
"name": "model.layers.21.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 25886976
},
{
"name": "model.layers.21.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 25887232
}
],
"md5sum": "7b562736630d98843ff9bfbcf21b3a17"
},
{
"dataPath": "params_shard_71.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.21.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "66f23a35335f95bf52a8a48e38726e23"
},
{
"dataPath": "params_shard_72.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.22.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "c10b785d79d403bf76152a0ce473ad90"
},
{
"dataPath": "params_shard_73.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.22.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "6973721e4b25169ee8a4431892ead8d1"
},
{
"dataPath": "params_shard_74.bin",
"format": "raw-shard",
"nbytes": 25917696,
"records": [
{
"name": "model.layers.21.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.21.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.21.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.21.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.21.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.21.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 20326656
},
{
"name": "model.layers.22.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 20336896
},
{
"name": "model.layers.22.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 20347136
}
],
"md5sum": "b31312ecdf00e25606851a20c0fab337"
},
{
"dataPath": "params_shard_75.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.22.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.22.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.22.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.22.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.22.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "7abd4eda628614b6030b4e1d43b418e8"
},
{
"dataPath": "params_shard_76.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.23.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "58021decfe4fdf28ee0d7ad1cf0f5490"
},
{
"dataPath": "params_shard_77.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.23.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "6eeb0ce332f010183a799255e1ce447b"
},
{
"dataPath": "params_shard_78.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.23.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "86d6727b6b4fc6dfba93ca811eb6eb59"
},
{
"dataPath": "params_shard_79.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.22.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.22.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.22.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.23.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.23.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.23.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.23.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.23.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "6f92e25017a8791ad87011a8ede4f153"
},
{
"dataPath": "params_shard_80.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.24.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "0f975139ecbdf7d2c73457e746a7fa45"
},
{
"dataPath": "params_shard_81.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.24.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "f7fc2a9de34410088e5f275961ba4198"
},
{
"dataPath": "params_shard_82.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.23.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.23.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.23.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.23.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.24.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.24.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "756c32631ca841ca4c75b4aeef899e5b"
},
{
"dataPath": "params_shard_83.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.24.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.24.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.24.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.24.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.24.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "7a5004bd79fa611d587e18ae39f3ab97"
},
{
"dataPath": "params_shard_84.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.25.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "acb56fa59b154bc9a4475680d18dc0ce"
},
{
"dataPath": "params_shard_85.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.25.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "f6f81a9189cbcbd9823de5947253c6e3"
},
{
"dataPath": "params_shard_86.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.25.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "c25f2cfa6f7fde5bbf164286744e26a9"
},
{
"dataPath": "params_shard_87.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.24.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.24.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.24.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.25.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.25.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.25.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.25.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.25.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "bb7431da7272077114995e05e17c27cb"
},
{
"dataPath": "params_shard_88.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.26.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "31e3012684d2275aa7ccfcc80817ffe2"
},
{
"dataPath": "params_shard_89.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.26.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "06812f672a727f248fa3daeed8fd86a2"
},
{
"dataPath": "params_shard_90.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.25.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.25.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.25.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.25.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.26.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.26.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "ca44de2ca83d18998850a16855682f9b"
},
{
"dataPath": "params_shard_91.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.26.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.26.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.26.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.26.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.26.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "f9116fc47ef4acf945e8bd37e0d03ed0"
},
{
"dataPath": "params_shard_92.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.27.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "d161459927c6278bd62676f41c8104df"
},
{
"dataPath": "params_shard_93.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.27.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "1b09e0f85368700f5b2e6a2738780a21"
},
{
"dataPath": "params_shard_94.bin",
"format": "raw-shard",
"nbytes": 28180992,
"records": [
{
"name": "model.layers.26.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.26.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.26.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.27.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 14745856
},
{
"name": "model.layers.27.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 25886976
},
{
"name": "model.layers.27.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 25887232
}
],
"md5sum": "60f7f1ed3963c4e38c2d896c959daf45"
},
{
"dataPath": "params_shard_95.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.27.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "8bb9c1f5e8fe5a2b3cbb9c71a7ec3a04"
},
{
"dataPath": "params_shard_96.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.28.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "7a38ad477fd17b35b5529454413e8b7f"
},
{
"dataPath": "params_shard_97.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.28.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "6328f50a2da3293f294832e210d9dc9a"
},
{
"dataPath": "params_shard_98.bin",
"format": "raw-shard",
"nbytes": 25917696,
"records": [
{
"name": "model.layers.27.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.27.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.27.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.27.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.27.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.27.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 20326656
},
{
"name": "model.layers.28.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 20336896
},
{
"name": "model.layers.28.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 20347136
}
],
"md5sum": "941af4223ff7a03445d3359a65165e7d"
},
{
"dataPath": "params_shard_99.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.28.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.28.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.28.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.28.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.28.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "eb6ce902171d2a7123c894ab80a9129b"
},
{
"dataPath": "params_shard_100.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.29.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "0eb03b8c3415ec6f9a64c5c723fa36cb"
},
{
"dataPath": "params_shard_101.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.29.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "16110c267dfe31749915b54833342ccc"
},
{
"dataPath": "params_shard_102.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.29.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "59817793a027a1ff264ae8d6c54192c6"
},
{
"dataPath": "params_shard_103.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.28.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.28.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.28.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.29.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.29.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.29.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.29.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.29.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "64a1be72aadff3480db2dbb556ffcadd"
},
{
"dataPath": "params_shard_104.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.30.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "45e5c079f881d214edbff3ba69c550b5"
},
{
"dataPath": "params_shard_105.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.30.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "20d41f0f33ce9b20a1c91a1650087037"
},
{
"dataPath": "params_shard_106.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.29.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.29.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.29.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.29.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.30.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.30.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "e6b8e5d8bd40b8454d306757f42a8778"
},
{
"dataPath": "params_shard_107.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.30.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.30.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.30.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.30.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.30.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "82f3dda32c42facc4fcfbde3a993eea8"
},
{
"dataPath": "params_shard_108.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.31.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "dd88258d97fd9b80842c7de5807883c4"
},
{
"dataPath": "params_shard_109.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.31.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "5bd21a22079aa2d477454b34c925ecb8"
},
{
"dataPath": "params_shard_110.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.31.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "94639fe21aa64515ec51d40d466589a0"
},
{
"dataPath": "params_shard_111.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.30.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.30.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.30.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.31.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.31.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.31.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.31.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.31.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "a00e653a2625cbf561e01036eae36d32"
},
{
"dataPath": "params_shard_112.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.32.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "70f3bccd4282c8f1272edab677abcce2"
},
{
"dataPath": "params_shard_113.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.32.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "82fc444531da3e621770367e05845c8d"
},
{
"dataPath": "params_shard_114.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.31.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.31.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.31.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.31.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.32.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.32.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "e61dbfe4d01c41e9d7a611db315d4d55"
},
{
"dataPath": "params_shard_115.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.32.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.32.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.32.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.32.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.32.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "1e9db11453a29820058ff754af300fb2"
},
{
"dataPath": "params_shard_116.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.33.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "d46ee8408ba1633959e403690c9c19b3"
},
{
"dataPath": "params_shard_117.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.33.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "ed8ff3d32e361b3ab681b04cf3d14d62"
},
{
"dataPath": "params_shard_118.bin",
"format": "raw-shard",
"nbytes": 28180992,
"records": [
{
"name": "model.layers.32.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.32.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.32.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.33.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 14745856
},
{
"name": "model.layers.33.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 25886976
},
{
"name": "model.layers.33.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 25887232
}
],
"md5sum": "87c826f15839a414d41aab90a0d92e97"
},
{
"dataPath": "params_shard_119.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.3.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "97228ba3a90ef4b28232ae09582e2509"
},
{
"dataPath": "params_shard_120.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.4.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "4ced40df5897c3cea9474e48460f025e"
},
{
"dataPath": "params_shard_121.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.4.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "8bb88c9639b37f2d291d2d4696594cce"
},
{
"dataPath": "params_shard_122.bin",
"format": "raw-shard",
"nbytes": 25917696,
"records": [
{
"name": "model.layers.33.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.33.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.33.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.3.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.3.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.3.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 20326656
},
{
"name": "model.layers.4.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 20336896
},
{
"name": "model.layers.4.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 20347136
}
],
"md5sum": "1c6130edf5a9e91cca78462a642d395d"
},
{
"dataPath": "params_shard_123.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.4.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.4.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.4.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.4.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.4.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "7f16a70eff218e4a3a2e4af9b79a3527"
},
{
"dataPath": "params_shard_124.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.5.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "b2f1f657d59bb259c50f9c9e58500650"
},
{
"dataPath": "params_shard_125.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "ce6025927293fa89b3dce864258e6fff"
},
{
"dataPath": "params_shard_126.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.5.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "706cd76aeda32378554b70fb7c626585"
},
{
"dataPath": "params_shard_127.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.4.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.4.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.4.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.5.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.5.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.5.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.5.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.5.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "204b122267f501eeb5397e52ee048393"
},
{
"dataPath": "params_shard_128.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.6.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "af768c95609a6bc5f3850b687b9f5b24"
},
{
"dataPath": "params_shard_129.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.6.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "0c01ee28ace107cabcbf92eaea25e340"
},
{
"dataPath": "params_shard_130.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.5.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.5.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.5.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.5.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.6.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.6.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "d2979b7528392e7c9e832753a594a93f"
},
{
"dataPath": "params_shard_131.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.6.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.6.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.6.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.6.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.6.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "1aa4ac2de2697ff55eecc34b524de67a"
},
{
"dataPath": "params_shard_132.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.7.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "9784d0e8acc0cd1b47631b9a00b21abc"
},
{
"dataPath": "params_shard_133.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.7.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "fc8765b42131a9b12049ae423e39f47c"
},
{
"dataPath": "params_shard_134.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.7.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "23335c65478f31d20ddc25536c6998c0"
},
{
"dataPath": "params_shard_135.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.6.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.6.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.6.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.7.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.7.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.7.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.7.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.7.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "6dc81ee9676dc07dfac1ccc23655eb4e"
},
{
"dataPath": "params_shard_136.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.8.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "7f1411d65aec001ee5ffb2bdfcc48268"
},
{
"dataPath": "params_shard_137.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.8.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "4c87415d441c1671bb701b67b36d9bd1"
},
{
"dataPath": "params_shard_138.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.7.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.7.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.7.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.7.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.8.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.8.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "19b7dd4f93c3324f31df976ea010206d"
},
{
"dataPath": "params_shard_139.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.8.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.8.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.8.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.8.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.8.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "f5e69929ed502e5cde482b40a3d5786e"
},
{
"dataPath": "params_shard_140.bin",
"format": "raw-shard",
"nbytes": 33096192,
"records": [
{
"name": "model.layers.8.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.8.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.8.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.9.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745856
},
{
"name": "model.layers.9.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 14746112
}
],
"md5sum": "6800a65754140a76b38f9de6a1553ace"
},
{
"dataPath": "params_shard_141.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.33.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "1df881e5873f3594bc4769cc48f5ca6c"
},
{
"dataPath": "params_shard_142.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.34.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "ec1a545baa68ff3ac968a3a1520e6819"
},
{
"dataPath": "params_shard_143.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.34.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "245d2d12c6875c8bbf526ef6109e9ea0"
},
{
"dataPath": "params_shard_144.bin",
"format": "raw-shard",
"nbytes": 28211456,
"records": [
{
"name": "model.layers.9.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.9.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.9.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.9.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.33.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.33.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
},
{
"name": "model.layers.33.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 22620416
},
{
"name": "model.layers.34.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 22630656
},
{
"name": "model.layers.34.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 22640896
}
],
"md5sum": "a372036726aad1a3305e0a1858d219b5"
},
{
"dataPath": "params_shard_145.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.34.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.34.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.34.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.34.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.34.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "075fd17c598e381abd993dd1b576d003"
},
{
"dataPath": "params_shard_146.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.35.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "c6453a852df4dd99dc8eb00766cdadd8"
},
{
"dataPath": "params_shard_147.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.35.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "7cbd403d9cbb43e86df93857ced9af3e"
},
{
"dataPath": "params_shard_148.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.35.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "5171dd7f22b490af5b980c55117ad2a4"
},
{
"dataPath": "params_shard_149.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.34.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.34.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.34.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.35.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.35.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.35.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.35.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.35.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "f7ea3cbd3a2c3ce07893086df83700eb"
},
{
"dataPath": "params_shard_150.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.36.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "2b603fc04ebfb2b0b13718f52e64b27c"
},
{
"dataPath": "params_shard_151.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.36.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "1668956662a354370ab2aa0016e96b12"
},
{
"dataPath": "params_shard_152.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.35.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.35.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.35.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.35.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.36.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.36.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "9cb4793b36b101e512e4037e60086969"
},
{
"dataPath": "params_shard_153.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.36.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.36.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.36.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.36.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.36.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "56534776ba94fb9c0de9f4bd48b6e0cc"
},
{
"dataPath": "params_shard_154.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.37.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "ae8920b97a845a40e90795ea45bee30b"
},
{
"dataPath": "params_shard_155.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.37.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "deefa837c6f0803c9925d58053cca38b"
},
{
"dataPath": "params_shard_156.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.37.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "be5984fbd562161ae5fa45aee580a677"
},
{
"dataPath": "params_shard_157.bin",
"format": "raw-shard",
"nbytes": 31478272,
"records": [
{
"name": "model.layers.36.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.36.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.36.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.37.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745856
},
{
"name": "model.layers.37.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 14756096
},
{
"name": "model.layers.37.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 20326656
},
{
"name": "model.layers.37.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 31467776
},
{
"name": "model.layers.37.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 31478016
}
],
"md5sum": "6b0daf2eec5d01ee4bf8f27b22467aa5"
},
{
"dataPath": "params_shard_158.bin",
"format": "raw-shard",
"nbytes": 44564480,
"records": [
{
"name": "model.layers.38.mlp.down_proj.q_weight",
"shape": [
5120,
2176
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 44564480,
"byteOffset": 0
}
],
"md5sum": "8d1cc6c18964b31285758596a82e804f"
},
{
"dataPath": "params_shard_159.bin",
"format": "raw-shard",
"nbytes": 89128960,
"records": [
{
"name": "model.layers.38.mlp.gate_up_proj.q_weight",
"shape": [
34816,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 89128960,
"byteOffset": 0
}
],
"md5sum": "fc3f2cdd0494055c09adbe06b8a6871c"
},
{
"dataPath": "params_shard_160.bin",
"format": "raw-shard",
"nbytes": 22620416,
"records": [
{
"name": "model.layers.37.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.37.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.37.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.37.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
},
{
"name": "model.layers.38.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039616
},
{
"name": "model.layers.38.mlp.down_proj.q_scale",
"shape": [
5120,
544
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 5570560,
"byteOffset": 17049856
}
],
"md5sum": "bd2842b1890044283c97879d79429627"
},
{
"dataPath": "params_shard_161.bin",
"format": "raw-shard",
"nbytes": 31795456,
"records": [
{
"name": "model.layers.38.mlp.gate_up_proj.q_scale",
"shape": [
34816,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 11141120,
"byteOffset": 0
},
{
"name": "model.layers.38.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 11141120
},
{
"name": "model.layers.38.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 11151360
},
{
"name": "model.layers.38.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 11151616
},
{
"name": "model.layers.38.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 29501696
}
],
"md5sum": "5df96e3a1463fc7961b9fe57faafbb17"
},
{
"dataPath": "params_shard_162.bin",
"format": "raw-shard",
"nbytes": 33096192,
"records": [
{
"name": "model.layers.38.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.38.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.38.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745600
},
{
"name": "model.layers.39.self_attn.k_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 14745856
},
{
"name": "model.layers.39.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 14746112
}
],
"md5sum": "29d729b9f2fdea204caed33bf3dfb6c1"
},
{
"dataPath": "params_shard_163.bin",
"format": "raw-shard",
"nbytes": 17039616,
"records": [
{
"name": "model.layers.39.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.39.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.39.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.39.self_attn.q_norm.weight",
"shape": [
128
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 256,
"byteOffset": 17039360
}
],
"md5sum": "17a57e7c1f639619fc9fb5432d71382d"
}
]
}