DeepCoder-14B-Preview-GPTQ-Int4 / ndarray-cache.json
numen-tech's picture
Add weights
6a9fe8c
{
"metadata": {
"ParamSize": 533,
"ParamBytes": 7617046528.0,
"BitsPerParam": 3.8073789790921078
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 389283840,
"records": [
{
"name": "lm_head.q_weight",
"shape": [
640,
152064
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 389283840,
"byteOffset": 0
}
],
"md5sum": "b4674daf61a60a9b559c512c324dfe33"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.46.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "7839fe93fe9de9247e9bb395640fc253"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.46.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "1e5c1d52466a5bcb52fe4a846340cf3d"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.47.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "5937eca514ebf8619935415a93d0b53d"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.47.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "34ff15ccbac9e354a06e9f271bb0362f"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.47.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "97eaffac050defd1dcfb85835048a380"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 389283840,
"records": [
{
"name": "model.embed_tokens.q_weight",
"shape": [
152064,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 389283840,
"byteOffset": 0
}
],
"md5sum": "61b02048cfa960b9e305043aaede305f"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 32956416,
"records": [
{
"name": "lm_head.q_scale",
"shape": [
40,
152064
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 12165120,
"byteOffset": 0
},
{
"name": "model.layers.46.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 12165120
},
{
"name": "model.layers.46.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 12175360
},
{
"name": "model.layers.46.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 13281280
},
{
"name": "model.layers.46.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 15493120
},
{
"name": "model.layers.47.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 15503360
},
{
"name": "model.layers.47.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 15513600
},
{
"name": "model.layers.47.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 16619520
},
{
"name": "model.layers.47.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 18831360
},
{
"name": "model.layers.47.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 18841600
},
{
"name": "model.layers.47.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 18855936
},
{
"name": "model.layers.47.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 19429376
},
{
"name": "model.layers.47.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32536576
},
{
"name": "model.norm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 32946176
}
],
"md5sum": "28af345c0a30992425762af71b3b4fc3"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.0.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "715a54d350f1cb60c6a93a70e14c6991"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.0.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "9b36105752dd33d69ce730957243176c"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.0.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "b6ccea183f1c72c6de025180ff5435b3"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.1.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "26edaebe65d5e75422743d5995935251"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.1.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "fbc49eef53537cf4279d66a2d9a1c40f"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 32407552,
"records": [
{
"name": "model.embed_tokens.q_scale",
"shape": [
152064,
40
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 12165120,
"byteOffset": 0
},
{
"name": "model.layers.0.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 12165120
},
{
"name": "model.layers.0.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 12175360
},
{
"name": "model.layers.0.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 13281280
},
{
"name": "model.layers.0.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 15493120
},
{
"name": "model.layers.0.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 15503360
},
{
"name": "model.layers.0.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 15517696
},
{
"name": "model.layers.0.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 16091136
},
{
"name": "model.layers.0.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 29198336
},
{
"name": "model.layers.1.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 29607936
},
{
"name": "model.layers.1.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 31819776
},
{
"name": "model.layers.1.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 31834112
}
],
"md5sum": "4889dacbcb77e6f1c2eb09e0e9e3d712"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.1.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "1fa819acf0582937fc8c3e24784aa9aa"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.2.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "6f1b1e29cac778242b9f57190a292642"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.2.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "92d484bc9f8898a8ec41bcba8527cbc2"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.2.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "14f4d3cc8195986e96ff4d74242b7244"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.3.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "fcebb8567e0f56026af778e8ac6e54b4"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.3.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "9cdd470733359cfd16caf18de3f61b66"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 33202176,
"records": [
{
"name": "model.layers.1.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.1.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.1.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.1.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.1.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14632960
},
{
"name": "model.layers.2.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14643200
},
{
"name": "model.layers.2.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 14653440
},
{
"name": "model.layers.2.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 15759360
},
{
"name": "model.layers.2.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17971200
},
{
"name": "model.layers.2.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 17981440
},
{
"name": "model.layers.2.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 17995776
},
{
"name": "model.layers.2.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 18569216
},
{
"name": "model.layers.2.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 31676416
},
{
"name": "model.layers.3.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 32086016
},
{
"name": "model.layers.3.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 32096256
}
],
"md5sum": "6c97c3af9f76bc1699aa3cf101ad9a0a"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.3.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.3.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.3.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.3.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.3.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "c3d1be35d36a35d418703db856e5b0e0"
},
{
"dataPath": "params_shard_22.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.4.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "5986b630ba2cd6f4423a8002c73629b4"
},
{
"dataPath": "params_shard_23.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.4.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "3ef16436bd1dbad39d34ed18c673739f"
},
{
"dataPath": "params_shard_24.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.4.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "8ceb2cf18bc97e95dbd3949f603b5b2b"
},
{
"dataPath": "params_shard_25.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.5.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "6c9dd46c89ec32c36246588d7335b9d6"
},
{
"dataPath": "params_shard_26.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "4ad09263b5cbce8de88eed57db11e652"
},
{
"dataPath": "params_shard_27.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.3.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.3.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.4.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.4.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.4.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.4.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.4.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.4.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.4.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.4.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.5.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.5.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "2c3b9d15fb0e09905bccd2ef369d0b87"
},
{
"dataPath": "params_shard_28.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.5.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.5.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.5.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.5.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "5dddcb193f67f02cba4a122e8d270f90"
},
{
"dataPath": "params_shard_29.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.10.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "c460fd2c88d211479512a723e8b86b5e"
},
{
"dataPath": "params_shard_30.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.10.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "fa0ff5587f86528c60af775f533feb55"
},
{
"dataPath": "params_shard_31.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.11.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "90c89f13fde6fe189208dbe1f0778bdd"
},
{
"dataPath": "params_shard_32.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.11.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "c0b7817b0084a2abaec13a79f8824405"
},
{
"dataPath": "params_shard_33.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.11.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "4c26f9b724a0765d93ba0b2a5f6c49eb"
},
{
"dataPath": "params_shard_34.bin",
"format": "raw-shard",
"nbytes": 20781056,
"records": [
{
"name": "model.layers.5.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.5.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.10.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.10.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.10.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.10.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.11.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16855040
},
{
"name": "model.layers.11.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 16865280
},
{
"name": "model.layers.11.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 17971200
},
{
"name": "model.layers.11.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 20183040
},
{
"name": "model.layers.11.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 20193280
},
{
"name": "model.layers.11.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20207616
}
],
"md5sum": "ad9de8ccfbb25681462173bd22efe975"
},
{
"dataPath": "params_shard_35.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.12.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "faed5a4d4f100271944a0a093d4fa67e"
},
{
"dataPath": "params_shard_36.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.12.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "4d4f853835d626b5a8acf859b6957ed7"
},
{
"dataPath": "params_shard_37.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.12.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "bef00a768845a859ba83b24d2e94efca"
},
{
"dataPath": "params_shard_38.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.13.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "5fe96d3000c636cec9106465196575f7"
},
{
"dataPath": "params_shard_39.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.13.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "db4f614a26ae2dbb8ec73bbc33cade43"
},
{
"dataPath": "params_shard_40.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.11.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.11.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.12.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.12.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.12.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.12.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.12.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.12.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.12.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.12.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.13.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.13.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "cb4697a264541eafac4ed20b770ccc26"
},
{
"dataPath": "params_shard_41.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.13.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.13.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.13.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.13.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.13.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "4c89ade7a4d3597d8bf83ffb08beed74"
},
{
"dataPath": "params_shard_42.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.14.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "492786998ec2ec4fbbf5925062f90358"
},
{
"dataPath": "params_shard_43.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.14.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "e6d488c803dce1d55b11b609022141a7"
},
{
"dataPath": "params_shard_44.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.14.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "1228c0d9fa103ae49d4cc628242a1fa6"
},
{
"dataPath": "params_shard_45.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.10.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "7c173fdd1a887a1bb44fbd7935588687"
},
{
"dataPath": "params_shard_46.bin",
"format": "raw-shard",
"nbytes": 31547392,
"records": [
{
"name": "model.layers.13.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.13.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.14.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.14.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.14.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.14.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.14.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.14.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.14.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.14.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.10.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 30959616
},
{
"name": "model.layers.10.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 30973952
}
],
"md5sum": "cc3f1bd082f1bb19977698dd3ac77f8f"
},
{
"dataPath": "params_shard_47.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.6.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "4dfbae196ef1b06cccb0e8ab6eddeb35"
},
{
"dataPath": "params_shard_48.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.6.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "3039ccf4af87c3c85db67e6135271f50"
},
{
"dataPath": "params_shard_49.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.6.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "848453f02b28f01e3bf5bbb3c5d19b25"
},
{
"dataPath": "params_shard_50.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.7.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "a837a4d63ffbbe1cb300e2c71dcc94b9"
},
{
"dataPath": "params_shard_51.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.7.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "452133efdd138c7ebcc285681aaaa7fc"
},
{
"dataPath": "params_shard_52.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.10.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.10.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.6.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.6.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.6.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.6.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.6.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.6.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.6.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.6.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.7.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.7.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "a41c5ea9a5bbe4b89a407d4ce2fa1716"
},
{
"dataPath": "params_shard_53.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.7.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.7.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.7.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.7.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.7.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "7584cf7539d976b10f74effd6112a805"
},
{
"dataPath": "params_shard_54.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.8.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "dbf638cb5f931887818a56bc5b762bb3"
},
{
"dataPath": "params_shard_55.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.8.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "8f582efa6c0afd924be63043afe967b3"
},
{
"dataPath": "params_shard_56.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.8.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "a550c72758ece727c1c540c9eae971ec"
},
{
"dataPath": "params_shard_57.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.9.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "ca7f3b2cbc7dba36a3984ac950def79a"
},
{
"dataPath": "params_shard_58.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.9.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "52468dac20be88a32edad15106678de1"
},
{
"dataPath": "params_shard_59.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.7.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.7.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.8.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.8.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.8.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.8.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.8.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.8.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.8.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.8.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.9.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.9.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "f9ea59d878146226b73dcd3fb38e860e"
},
{
"dataPath": "params_shard_60.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.9.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.9.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.9.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.9.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.9.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "3e8a4ea076f69e6c4aa988f0085ef139"
},
{
"dataPath": "params_shard_61.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.15.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "fa6bfa954bc21f5d9c6d3d84acd4e37c"
},
{
"dataPath": "params_shard_62.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.15.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "b256a0c6012783b0c5be3c8965ef5741"
},
{
"dataPath": "params_shard_63.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.15.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "7fa789cd6f65193083666f61e30735af"
},
{
"dataPath": "params_shard_64.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.16.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "88fb5dd23841b4b726a67d471c66b677"
},
{
"dataPath": "params_shard_65.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.16.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "d49e90607ee038148fc449638c864f7a"
},
{
"dataPath": "params_shard_66.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.9.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.9.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.15.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.15.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.15.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.15.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.15.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.15.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.15.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.15.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.16.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.16.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "013d6301c46b6b5d5725106b5e1a91c5"
},
{
"dataPath": "params_shard_67.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.16.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.16.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.16.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.16.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.16.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "937f191c319ce0d241822ebba60da73c"
},
{
"dataPath": "params_shard_68.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.17.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "696d359df81a295a51d0d0f8baa02655"
},
{
"dataPath": "params_shard_69.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.17.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "163c7a37174c0b01b80fc29fb5227344"
},
{
"dataPath": "params_shard_70.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.17.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "ceb724796272bf748ceed7cadd77ff1f"
},
{
"dataPath": "params_shard_71.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.18.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "13334c8fa552281332d316c2d95a8f54"
},
{
"dataPath": "params_shard_72.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.18.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "8d8edd7b7715aa1f7e837ef7233108da"
},
{
"dataPath": "params_shard_73.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.16.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.16.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.17.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.17.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.17.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.17.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.17.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.17.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.17.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.17.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.18.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.18.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "407225ffb87f1ee5131e2c26ce6146d4"
},
{
"dataPath": "params_shard_74.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.18.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.18.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.18.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.18.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.18.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "5cff06333e005ac82abbec2afb2b48d5"
},
{
"dataPath": "params_shard_75.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.19.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "4f5a69087e87495f25cb9340635c2492"
},
{
"dataPath": "params_shard_76.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.19.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "0198fc8f45475a34335b6d5751db99be"
},
{
"dataPath": "params_shard_77.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.19.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "6b2b0b7fcf767ea9f1769db6b00ba35e"
},
{
"dataPath": "params_shard_78.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.20.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "5d6882c383d15e57c2d07f3d1453108d"
},
{
"dataPath": "params_shard_79.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.20.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "afaea38689db3006429048f536fdac72"
},
{
"dataPath": "params_shard_80.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.18.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.18.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.19.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 13516800
},
{
"name": "model.layers.19.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 15728640
},
{
"name": "model.layers.19.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 15742976
},
{
"name": "model.layers.19.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 16316416
},
{
"name": "model.layers.19.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 29423616
},
{
"name": "model.layers.19.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 29833216
},
{
"name": "model.layers.19.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 29843456
},
{
"name": "model.layers.19.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30949376
},
{
"name": "model.layers.20.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.20.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "de8ed83879133b0eb5ef35b24911eb03"
},
{
"dataPath": "params_shard_81.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.20.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.20.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.20.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.20.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.20.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "23d35122dc691939160415449bdbed59"
},
{
"dataPath": "params_shard_82.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.21.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "bcd3cb90aa300808d95fe11728eb9b50"
},
{
"dataPath": "params_shard_83.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.21.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "1e699e011781e44b9345de73991a7da0"
},
{
"dataPath": "params_shard_84.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.21.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "0c5c63d447b5c41e392b2b9a67763fff"
},
{
"dataPath": "params_shard_85.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.22.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "c2c2797f69c8a472a059f1a7c904e7b1"
},
{
"dataPath": "params_shard_86.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.22.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "e92e0744bc889984c1d26dff4a1234d2"
},
{
"dataPath": "params_shard_87.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.20.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.20.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.21.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.21.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.21.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.21.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.21.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.21.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.21.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.21.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.22.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.22.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "4bffeb3ea709bd3d72e3de5b5f0a8f1e"
},
{
"dataPath": "params_shard_88.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.22.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.22.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.22.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.22.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.22.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "d496a370828f53c448a5647aea0366d1"
},
{
"dataPath": "params_shard_89.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.23.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "abcaea7c0dad266379740d7b21d989d2"
},
{
"dataPath": "params_shard_90.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.23.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "dac82b3fd1b782c47632dd01b92a1eee"
},
{
"dataPath": "params_shard_91.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.23.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "36b7b156ae74ce4ac7fed7e0bbf303bb"
},
{
"dataPath": "params_shard_92.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.24.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "ad08758b259ce8a60aab2f90f4aacfe9"
},
{
"dataPath": "params_shard_93.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.24.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "655264c1e9fe3cdf4b60b1cdd983c2e7"
},
{
"dataPath": "params_shard_94.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.22.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.22.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.23.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.23.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.23.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.23.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.23.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.23.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.23.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.23.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.24.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.24.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "f050803c546b2b89e27db57dbce6ae45"
},
{
"dataPath": "params_shard_95.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.24.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.24.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.24.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.24.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.24.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "18d8a560523e9f4a800bfcd89dd6f31e"
},
{
"dataPath": "params_shard_96.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.25.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "7a4f26b4999cf4604d46e9e5ba5adae1"
},
{
"dataPath": "params_shard_97.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.25.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "23cb6be35935cee0a419692f06e22623"
},
{
"dataPath": "params_shard_98.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.25.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "90fb54b71833747de9af16c5e40c4e1f"
},
{
"dataPath": "params_shard_99.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.26.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "af2989046482d2c8dc9cdd6108b8d12a"
},
{
"dataPath": "params_shard_100.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.26.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "d698b9332c2e3c31800780f414a37fc1"
},
{
"dataPath": "params_shard_101.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.24.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.24.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.25.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.25.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.25.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.25.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.25.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.25.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.25.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.25.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.26.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.26.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "1269e68c03a108126e684a0709c183af"
},
{
"dataPath": "params_shard_102.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.26.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.26.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.26.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.26.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.26.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "9681dccb4240835f1b39aaff8aa8298e"
},
{
"dataPath": "params_shard_103.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.27.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "880221a035d99bcb45f2ec4276993c44"
},
{
"dataPath": "params_shard_104.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.27.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "b9965b853dc02f57681969b6902dc032"
},
{
"dataPath": "params_shard_105.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.27.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "6f7d6fe3ce1c08467f4ac71c26e42b82"
},
{
"dataPath": "params_shard_106.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.28.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "f26c5775527ed24bd5ee9d0db4acfc77"
},
{
"dataPath": "params_shard_107.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.28.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "a112c4cc77b7ad1a9000dbe41dce0366"
},
{
"dataPath": "params_shard_108.bin",
"format": "raw-shard",
"nbytes": 33185792,
"records": [
{
"name": "model.layers.26.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.26.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.27.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.27.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.27.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.27.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.27.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.27.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.27.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.27.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.28.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 30959616
},
{
"name": "model.layers.28.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 33171456
}
],
"md5sum": "9cb68b679b77425953291111e784d941"
},
{
"dataPath": "params_shard_109.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.28.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "4f0bff19323645301ad2c018ad79a134"
},
{
"dataPath": "params_shard_110.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.29.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "2879e9370c79d7fc00ab52d37103af03"
},
{
"dataPath": "params_shard_111.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.29.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "0a34ad92e46a800e7677b89a0bc42ab1"
},
{
"dataPath": "params_shard_112.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.29.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "59e305045996faddc43b13d1db45d3de"
},
{
"dataPath": "params_shard_113.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.30.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "2e634693a7f9e111649b9c80fbe0bdb1"
},
{
"dataPath": "params_shard_114.bin",
"format": "raw-shard",
"nbytes": 32669696,
"records": [
{
"name": "model.layers.28.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 0
},
{
"name": "model.layers.28.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 573440
},
{
"name": "model.layers.28.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13680640
},
{
"name": "model.layers.28.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14090240
},
{
"name": "model.layers.28.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 14100480
},
{
"name": "model.layers.28.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 15206400
},
{
"name": "model.layers.29.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 15216640
},
{
"name": "model.layers.29.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 15226880
},
{
"name": "model.layers.29.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 16332800
},
{
"name": "model.layers.29.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 18544640
},
{
"name": "model.layers.29.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 18554880
},
{
"name": "model.layers.29.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 18569216
},
{
"name": "model.layers.29.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 19142656
},
{
"name": "model.layers.29.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 32249856
},
{
"name": "model.layers.30.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 32659456
}
],
"md5sum": "a83966e4728c399935ba9334dfbb16f8"
},
{
"dataPath": "params_shard_115.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.30.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "c137056d4bfe3b27e9f7a9f738c50aff"
},
{
"dataPath": "params_shard_116.bin",
"format": "raw-shard",
"nbytes": 22265856,
"records": [
{
"name": "model.layers.30.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 0
},
{
"name": "model.layers.30.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 1105920
},
{
"name": "model.layers.30.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 3317760
},
{
"name": "model.layers.30.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 3328000
},
{
"name": "model.layers.30.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 3342336
},
{
"name": "model.layers.30.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 21692416
}
],
"md5sum": "36191bedd1ae7ccaed476bbdd08c37e1"
},
{
"dataPath": "params_shard_117.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.31.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "011d06cdf298f2884bbc7d85b8d27c59"
},
{
"dataPath": "params_shard_118.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.31.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "fb59e7c7550a9bae6778cffa21e30c32"
},
{
"dataPath": "params_shard_119.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.31.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "d4499e09b631b7085481e1268ef9e79e"
},
{
"dataPath": "params_shard_120.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.32.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "31f441bc0af0dfbd01d406d8e3c84f66"
},
{
"dataPath": "params_shard_121.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.32.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "68e4cbf95aee91da862dc04129155d84"
},
{
"dataPath": "params_shard_122.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.30.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.30.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.31.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.31.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.31.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.31.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.31.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.31.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.31.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.31.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.32.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.32.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "0ae05e7bd80906c1b34d78d50dccefe0"
},
{
"dataPath": "params_shard_123.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.32.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.32.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.32.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.32.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.32.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "addeb39a6e2e3f72a929d36ddcd6848d"
},
{
"dataPath": "params_shard_124.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.33.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "d4f9a765fe8d70a7de03fbc71945f611"
},
{
"dataPath": "params_shard_125.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.33.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "8ebade09312c7964880728fd5c74fae2"
},
{
"dataPath": "params_shard_126.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.33.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "c297acf8b0672b7c7124cd610f93c8e9"
},
{
"dataPath": "params_shard_127.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.34.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "153c025583f0468bcbedfcf585504063"
},
{
"dataPath": "params_shard_128.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.34.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "4852963968808c8ec503ae7977fef1e7"
},
{
"dataPath": "params_shard_129.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.32.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.32.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.33.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.33.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.33.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.33.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.33.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.33.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.33.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.33.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.34.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.34.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "70493b0efc9eae1b3a9455dd2868b58b"
},
{
"dataPath": "params_shard_130.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.34.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.34.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.34.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.34.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.34.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "48649b63d3be926a45dec661c9a259c2"
},
{
"dataPath": "params_shard_131.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.35.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "de145f136663a8b5f36dad7169ffba40"
},
{
"dataPath": "params_shard_132.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.35.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "37e48d85c97a3ae7c87c40bcc18e9732"
},
{
"dataPath": "params_shard_133.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.35.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "312f51e89ac100e17335937e59bd9e08"
},
{
"dataPath": "params_shard_134.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.36.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "f6e55d268fff1481c36018080eabd708"
},
{
"dataPath": "params_shard_135.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.36.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "7328788b5b086355c5a4deaf10e63b98"
},
{
"dataPath": "params_shard_136.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.34.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.34.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.35.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.35.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.35.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.35.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.35.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.35.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.35.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.35.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.36.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.36.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "1cbdd822db3b9494d2ac11008716f8df"
},
{
"dataPath": "params_shard_137.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.36.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.36.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.36.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.36.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.36.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "05e76972f99b2ef47e7bd480b781ed16"
},
{
"dataPath": "params_shard_138.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.37.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "fdb6c093498a53f78c357132467d0b23"
},
{
"dataPath": "params_shard_139.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.37.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "f93d716ca68a0e021b86a6d783e81809"
},
{
"dataPath": "params_shard_140.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.37.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "db4dfbc805f84e94a66e129e2190a6bb"
},
{
"dataPath": "params_shard_141.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.38.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "b01f08a3edfee298f0cd683bfd006b0d"
},
{
"dataPath": "params_shard_142.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.38.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "8565b05bfe652ab44c1f7911dba84c29"
},
{
"dataPath": "params_shard_143.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.36.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.36.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.37.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 13516800
},
{
"name": "model.layers.37.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 15728640
},
{
"name": "model.layers.37.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 15742976
},
{
"name": "model.layers.37.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 16316416
},
{
"name": "model.layers.37.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 29423616
},
{
"name": "model.layers.37.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 29833216
},
{
"name": "model.layers.37.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 29843456
},
{
"name": "model.layers.37.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30949376
},
{
"name": "model.layers.38.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.38.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "50a81acb3e1f9f1f0b4255550a4a4d21"
},
{
"dataPath": "params_shard_144.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.38.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.38.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.38.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.38.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.38.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "c45a2397c04173b59f04ffcd77c6807e"
},
{
"dataPath": "params_shard_145.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.39.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "00b06aa4bc619a1e2df22aa7b4cef0c4"
},
{
"dataPath": "params_shard_146.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.39.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "a65c0d7058a35b7f61f09a2825548835"
},
{
"dataPath": "params_shard_147.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.39.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "2095f5b2c28b6debf9be1cfd242843d0"
},
{
"dataPath": "params_shard_148.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.40.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "2918c0cf6ba95768fe9bc36097b961c5"
},
{
"dataPath": "params_shard_149.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.40.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "586ae65479fe5f83b410df7bdb16d329"
},
{
"dataPath": "params_shard_150.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.38.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.38.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.39.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.39.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.39.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.39.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.39.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.39.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.39.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.39.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.40.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.40.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "7009647a36fd9e754d06da6a3ffaa2dc"
},
{
"dataPath": "params_shard_151.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.40.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.40.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.40.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.40.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.40.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "2a799e69dca3706458696f53b781a346"
},
{
"dataPath": "params_shard_152.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.41.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "575600f7ca1be118190bdc05f248d5bd"
},
{
"dataPath": "params_shard_153.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.41.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "0e3958eb3bf3eceb02591b0601da9d7e"
},
{
"dataPath": "params_shard_154.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.41.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "77b5eb2bc198d7b96335ec260bbee630"
},
{
"dataPath": "params_shard_155.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.42.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "868db7fec66bc9039915af13cbce9de9"
},
{
"dataPath": "params_shard_156.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.42.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "7e8f7a8360452f622095e26ac90ed173"
},
{
"dataPath": "params_shard_157.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.40.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.40.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.41.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.41.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.41.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.41.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.41.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.41.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.41.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.41.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.42.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.42.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "cb4a22a758c109923d077fe9f963ab5c"
},
{
"dataPath": "params_shard_158.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.42.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.42.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.42.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.42.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.42.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "70819e15fdd98d783f7e1d3fb36f7b82"
},
{
"dataPath": "params_shard_159.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.43.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "3db794d1211729631113262258b9249b"
},
{
"dataPath": "params_shard_160.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.43.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "5065b83e254846dbf274fb0ede5152e4"
},
{
"dataPath": "params_shard_161.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.43.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "fcdcd68a15d1d94df47d6201cc7f7bf8"
},
{
"dataPath": "params_shard_162.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.44.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "0fc65f1dfd4189ab8ee1192909cf734a"
},
{
"dataPath": "params_shard_163.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.44.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "ac03316527818a6b3f7bdb8b32a398b4"
},
{
"dataPath": "params_shard_164.bin",
"format": "raw-shard",
"nbytes": 32075776,
"records": [
{
"name": "model.layers.42.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.42.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.43.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.43.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.43.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.43.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.43.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.43.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.43.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.43.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.44.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30959616
},
{
"name": "model.layers.44.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 30969856
}
],
"md5sum": "956ad1d2bd8130f3c4ab342023f00c07"
},
{
"dataPath": "params_shard_165.bin",
"format": "raw-shard",
"nbytes": 21159936,
"records": [
{
"name": "model.layers.44.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 0
},
{
"name": "model.layers.44.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 2211840
},
{
"name": "model.layers.44.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 2222080
},
{
"name": "model.layers.44.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 2236416
},
{
"name": "model.layers.44.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 20586496
}
],
"md5sum": "0b0218c25487b2484d5f1100a6566b02"
},
{
"dataPath": "params_shard_166.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.45.mlp.down_proj.q_weight",
"shape": [
1728,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "c7710fbd9c88e07aae3b86795b3a14b2"
},
{
"dataPath": "params_shard_167.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.45.mlp.gate_up_proj.q_weight",
"shape": [
640,
27648
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "609fe3f42cf4898e93c331f424aacdd4"
},
{
"dataPath": "params_shard_168.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.45.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "ab8705acee4d52ae8905c823bbae6652"
},
{
"dataPath": "params_shard_169.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.46.self_attn.c_attn.q_weight",
"shape": [
640,
7168
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "3c2a1f7ac352ad6cfde8b84e95e3593e"
},
{
"dataPath": "params_shard_170.bin",
"format": "raw-shard",
"nbytes": 31547392,
"records": [
{
"name": "model.layers.44.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.44.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
},
{
"name": "model.layers.45.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13516800
},
{
"name": "model.layers.45.mlp.down_proj.q_scale",
"shape": [
108,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1105920,
"byteOffset": 13527040
},
{
"name": "model.layers.45.mlp.gate_up_proj.q_scale",
"shape": [
40,
27648
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2211840,
"byteOffset": 14632960
},
{
"name": "model.layers.45.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 16844800
},
{
"name": "model.layers.45.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 16855040
},
{
"name": "model.layers.45.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 16869376
},
{
"name": "model.layers.45.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 17442816
},
{
"name": "model.layers.45.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 30550016
},
{
"name": "model.layers.46.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 30959616
},
{
"name": "model.layers.46.self_attn.c_attn.q_scale",
"shape": [
40,
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 573440,
"byteOffset": 30973952
}
],
"md5sum": "b309a22ce9f9e3b27aa1c379ff762c9a"
},
{
"dataPath": "params_shard_171.bin",
"format": "raw-shard",
"nbytes": 13516800,
"records": [
{
"name": "model.layers.46.self_attn.o_proj.q_weight",
"shape": [
640,
5120
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.46.self_attn.o_proj.q_scale",
"shape": [
40,
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 409600,
"byteOffset": 13107200
}
],
"md5sum": "ded0d827346eb598733b7aafea1ed5f4"
}
]
}