Llama-MLC_1 / ndarray-cache.json
HarshvardhanCn01's picture
Upload folder using huggingface_hub
bed342d verified
{
"metadata": {
"ParamSize": 163,
"ParamBytes": 695245056.0,
"BitsPerParam": 4.500628907887781
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 131336192,
"records": [
{
"name": "model.embed_tokens.q_weight",
"shape": [
128258,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 131336192,
"byteOffset": 0
}
],
"md5sum": "6124bad0e973c9ca2eb02850856705b9"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.0.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "73428a8effc04d761eed7840fdf6b8c7"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 31498496,
"records": [
{
"name": "model.embed_tokens.q_scale",
"shape": [
128258,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 16417024,
"byteOffset": 0
},
{
"name": "model.layers.0.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 16417024
},
{
"name": "model.layers.0.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 16421120
},
{
"name": "model.layers.0.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 24809728
},
{
"name": "model.layers.0.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 25858304
},
{
"name": "model.layers.0.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 27955456
},
{
"name": "model.layers.0.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 27959552
},
{
"name": "model.layers.0.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31105280
}
],
"md5sum": "a857d6062df7e6034a3f00428a6eba4a"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.0.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.0.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.1.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.1.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.1.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.1.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.1.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.1.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "35880f151a0320540423f6a56b0e39b0"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.1.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.1.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.1.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.1.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.10.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.10.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.10.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.10.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "0399e7018251ef272ce9e454be12e330"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.11.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "d16e6c34f09f4a237b34a03bb76865df"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.10.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.10.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.10.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.10.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.10.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.10.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.11.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.11.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.11.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.11.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.11.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.11.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.11.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.11.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.11.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.layers.12.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "89a1adc7f1d1810c4676d17d0438d17c"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 31854592,
"records": [
{
"name": "model.layers.12.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.12.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.12.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 9437184
},
{
"name": "model.layers.12.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 26214400
},
{
"name": "model.layers.12.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 28311552
},
{
"name": "model.layers.12.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 28315648
},
{
"name": "model.layers.12.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31461376
}
],
"md5sum": "580e966efc6dccccff02143a3ac6ea93"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.12.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.12.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.13.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.13.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.13.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.13.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.13.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.13.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "5d0ff86eb3582a01fc5cc8ff05994bac"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.13.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.13.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.13.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.13.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.14.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.14.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.14.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.14.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "1069869e7d8fc765ba31c9b2649c7e97"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.15.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "40bdf3661bdbdf6287d8e6b384a8106b"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.14.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.14.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.14.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.14.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.14.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.14.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.15.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.15.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.15.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.15.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.15.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.15.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.15.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.15.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.15.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.layers.2.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "99b309c374b5fcd2fa939c6823f16719"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 31854592,
"records": [
{
"name": "model.layers.2.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.2.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.2.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 9437184
},
{
"name": "model.layers.2.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 26214400
},
{
"name": "model.layers.2.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 28311552
},
{
"name": "model.layers.2.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 28315648
},
{
"name": "model.layers.2.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31461376
}
],
"md5sum": "cb9bd7c0736545c9df8195c45184af09"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.2.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.2.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.3.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.3.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.3.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.3.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.3.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.3.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "ad094798695790f88b895530893f4026"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.3.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.3.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.3.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.3.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.4.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.4.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.4.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.4.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "607563464d8a59cae7c60da0950c56a0"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "b86b212a485e75ab5577f7fce6f0e891"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.4.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.4.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.4.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.4.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.4.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.4.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.5.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.5.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.5.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.5.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.5.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.5.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.5.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.5.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.5.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.layers.6.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "c88417f337ecbaeb8a7ac02535e9c5a1"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 31854592,
"records": [
{
"name": "model.layers.6.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.6.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.6.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 9437184
},
{
"name": "model.layers.6.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 26214400
},
{
"name": "model.layers.6.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 28311552
},
{
"name": "model.layers.6.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 28315648
},
{
"name": "model.layers.6.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31461376
}
],
"md5sum": "4ceff3c4fa259163a596f446a34bfa90"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.6.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.6.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.7.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.7.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.7.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.7.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.7.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.7.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "ad8d370c9603a4cb7a23e17b3d8df3b7"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.7.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.7.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.7.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.7.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.8.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.8.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.8.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.8.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "23b562f2a370825fae321c560bb522b8"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.9.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "d4f7772740d5b196b4edb2389b91bfc1"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.8.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.8.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.8.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.8.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.8.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.8.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.9.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.9.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.9.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.9.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.9.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.9.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.9.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.9.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.9.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.norm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "af26897b30f986377a5cb178c8b1c09f"
}
]
}