Qwen2.5-14B-Instruct-q4f16_1-MLC / ndarray-cache.json
riczhou's picture
Upload folder using huggingface_hub
83e8ea7 verified
{
"metadata": {
"ParamSize": 533,
"ParamBytes": 8309352448.0,
"BitsPerParam": 4.50065457508222
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 389283840,
"records": [
{
"name": "lm_head.q_weight",
"shape": [
152064,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 389283840,
"byteOffset": 0
}
],
"md5sum": "c4045b24e2a5e3fa81b086242987c5f3"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 48660480,
"records": [
{
"name": "lm_head.q_scale",
"shape": [
152064,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 48660480,
"byteOffset": 0
}
],
"md5sum": "5c08d18c8fc389d879adf3fda35199f5"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.47.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "a26255a3bc7f917f96ec13fb52865f2c"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 389283840,
"records": [
{
"name": "model.embed_tokens.q_weight",
"shape": [
152064,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 389283840,
"byteOffset": 0
}
],
"md5sum": "64450a8b3575dca9612cc4c24a01280e"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 48660480,
"records": [
{
"name": "model.embed_tokens.q_scale",
"shape": [
152064,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 48660480,
"byteOffset": 0
}
],
"md5sum": "ab2e38a70c77cf70496930a9ba57e6ec"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.0.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "e116e839b0d2b6c7da2e14722dc656b9"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.0.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "f019c73573cc7fd6d416f9208cc30951"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.0.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "2bdb1ef96de5a1de09379aad1826e9a0"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 33140736,
"records": [
{
"name": "model.layers.47.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 0
},
{
"name": "model.norm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 4423680
},
{
"name": "model.layers.0.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 4433920
},
{
"name": "model.layers.0.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 4444160
},
{
"name": "model.layers.0.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 8867840
},
{
"name": "model.layers.0.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17715200
},
{
"name": "model.layers.0.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 17725440
},
{
"name": "model.layers.0.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 17739776
},
{
"name": "model.layers.0.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 20033536
}
],
"md5sum": "16929625efc4e35e38b033e66b32a896"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.1.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "d38c53b4b7bce6b3d938bcc183d406ee"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.1.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "4f9ebae9ac8a8654585461968fc91445"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 33294336,
"records": [
{
"name": "model.layers.0.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 0
},
{
"name": "model.layers.1.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 1638400
},
{
"name": "model.layers.1.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 1648640
},
{
"name": "model.layers.1.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 6072320
},
{
"name": "model.layers.1.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14919680
},
{
"name": "model.layers.1.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 14929920
},
{
"name": "model.layers.1.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 14944256
}
],
"md5sum": "3dd719881b0e33106e24449585268ff2"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.2.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "263e5a27cdba91411576e6d1ac2a1a2a"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.2.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "32b8fd0272dcec0cdae8c455c7fdaec2"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.2.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "544f4a9feab4385b54225e217dbc480a"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 32638976,
"records": [
{
"name": "model.layers.1.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.1.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.1.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.2.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039360
},
{
"name": "model.layers.2.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 17049600
},
{
"name": "model.layers.2.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 21473280
},
{
"name": "model.layers.2.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30320640
},
{
"name": "model.layers.2.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 30330880
},
{
"name": "model.layers.2.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 30345216
}
],
"md5sum": "d3bee97d7b887f0501ce63b5bb2fb6d3"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.3.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "6a8ab592fabe2d2eed49fe0bc7571fb5"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.3.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "7bfd064a78f0261af03614fcefef250f"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.3.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "8f6e6d55bf3786e7c6843529e1e8a175"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.2.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.2.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.3.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.3.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.3.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.3.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.3.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.3.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "bb893037a59833a87ab8838ad54790e0"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 33130496,
"records": [
{
"name": "model.layers.3.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.3.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.4.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.4.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14755840
},
{
"name": "model.layers.4.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 14766080
},
{
"name": "model.layers.4.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 14780416
}
],
"md5sum": "8a035f302771d845829b60b89b8f841e"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.10.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "99bd1b6fcf4288fe7e3e7aba7c1b14ac"
},
{
"dataPath": "params_shard_22.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.10.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "a896a7fa384a0dab9f495a3c387f1792"
},
{
"dataPath": "params_shard_23.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.10.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "c349e344a85f963ba5c69342cc58d118"
},
{
"dataPath": "params_shard_24.bin",
"format": "raw-shard",
"nbytes": 32638976,
"records": [
{
"name": "model.layers.4.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.4.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.4.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.10.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039360
},
{
"name": "model.layers.10.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 17049600
},
{
"name": "model.layers.10.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 21473280
},
{
"name": "model.layers.10.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30320640
},
{
"name": "model.layers.10.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 30330880
},
{
"name": "model.layers.10.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 30345216
}
],
"md5sum": "31f38af0281a487ff0f986330633b84e"
},
{
"dataPath": "params_shard_25.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.11.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "8edbb980e1d88d3b465964de9489bb22"
},
{
"dataPath": "params_shard_26.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.11.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "ea2436a2262d48d2d7b41234aefcc7d2"
},
{
"dataPath": "params_shard_27.bin",
"format": "raw-shard",
"nbytes": 25921536,
"records": [
{
"name": "model.layers.10.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.10.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.11.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.11.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 14755840
},
{
"name": "model.layers.11.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 23603200
},
{
"name": "model.layers.11.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 23613440
},
{
"name": "model.layers.11.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 23627776
}
],
"md5sum": "40534b6a4a641aaa9a16b9d4004c001d"
},
{
"dataPath": "params_shard_28.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.4.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "835c3355ab01a2605b86112232517d9e"
},
{
"dataPath": "params_shard_29.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.4.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "f1b7287ac4522fa45f26ac2002f22f2b"
},
{
"dataPath": "params_shard_30.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.5.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "9ca53871065a20f25f38040b903ce7d6"
},
{
"dataPath": "params_shard_31.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "57fab020613b657620053a410916e179"
},
{
"dataPath": "params_shard_32.bin",
"format": "raw-shard",
"nbytes": 32450560,
"records": [
{
"name": "model.layers.11.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.11.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.4.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14745600
},
{
"name": "model.layers.4.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19169280
},
{
"name": "model.layers.5.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28016640
},
{
"name": "model.layers.5.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 28026880
}
],
"md5sum": "9f877cd0f6682341f02ed7a375df37c6"
},
{
"dataPath": "params_shard_33.bin",
"format": "raw-shard",
"nbytes": 29515776,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 0
},
{
"name": "model.layers.5.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 8847360
},
{
"name": "model.layers.5.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 8857600
},
{
"name": "model.layers.5.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 8871936
},
{
"name": "model.layers.5.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 27222016
}
],
"md5sum": "c1b6f32c9d45c53c96baaea334b07655"
},
{
"dataPath": "params_shard_34.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.6.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "f16aaf33a11500e717cbb3e9476c9ba8"
},
{
"dataPath": "params_shard_35.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.6.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "b250f8e09d219cd1456a64cbf8b596e0"
},
{
"dataPath": "params_shard_36.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.6.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "0bf8e42035b1292714c7b0d0f65ddf37"
},
{
"dataPath": "params_shard_37.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.5.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.5.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.6.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.6.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.6.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.6.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.6.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.6.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "6ea5d70551ca979307af62f781d232e3"
},
{
"dataPath": "params_shard_38.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.7.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "2fd36e0ab7f1f64db3bc5685bca97fa2"
},
{
"dataPath": "params_shard_39.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.7.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "d231ad2c1ca78066783ccec86a180578"
},
{
"dataPath": "params_shard_40.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.7.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "6ffa4e1de9a9512a37f01073a5d02656"
},
{
"dataPath": "params_shard_41.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.6.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.6.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.7.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.7.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.7.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.7.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.7.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.7.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "80843ae5e9daed13bf6964c875a6a1d3"
},
{
"dataPath": "params_shard_42.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.8.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "33e34437816a18f732826afbe426ab99"
},
{
"dataPath": "params_shard_43.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.8.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "ac9934b60c80b247e63664fcb4f860db"
},
{
"dataPath": "params_shard_44.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.8.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "c81625bb0b75bd4d45fab0396748775c"
},
{
"dataPath": "params_shard_45.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.7.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.7.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.8.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.8.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.8.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.8.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.8.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.8.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "fa23a48395ecabee9f0fe758b8d63289"
},
{
"dataPath": "params_shard_46.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.9.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "5aa545bcd1c4beaa776872576fa061cf"
},
{
"dataPath": "params_shard_47.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.9.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "e6c96638e91e4e7935008354d827b7ef"
},
{
"dataPath": "params_shard_48.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.9.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "4bbc322edc60e0085fceb6b571abf8b8"
},
{
"dataPath": "params_shard_49.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.8.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.8.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.9.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.9.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.9.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.9.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.9.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.9.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "27cbf0128e1576a748a51a5beb1ea30d"
},
{
"dataPath": "params_shard_50.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.11.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "2183e18fe108b581e4211accc54f4801"
},
{
"dataPath": "params_shard_51.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.12.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "570320057b5e9e8b948137258ef693ba"
},
{
"dataPath": "params_shard_52.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.12.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "31a08e8f92cb10bbe37803b97a3a5c69"
},
{
"dataPath": "params_shard_53.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.12.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "1bb6ca80469cd0bfa0879c175b7fdc55"
},
{
"dataPath": "params_shard_54.bin",
"format": "raw-shard",
"nbytes": 32475136,
"records": [
{
"name": "model.layers.9.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.9.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.11.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14745600
},
{
"name": "model.layers.12.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 19169280
},
{
"name": "model.layers.12.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 19179520
},
{
"name": "model.layers.12.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 23603200
},
{
"name": "model.layers.12.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 32450560
},
{
"name": "model.layers.12.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 32460800
}
],
"md5sum": "b0dbab756a67816bd8fd42dcb7f0e30a"
},
{
"dataPath": "params_shard_55.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.13.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "5591efc277bae02a75fa3a04ffe9af25"
},
{
"dataPath": "params_shard_56.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.13.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "8febb6be76c5ba3df5e648515622b771"
},
{
"dataPath": "params_shard_57.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.13.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "f2e1199f436002dc2669f11435f3a949"
},
{
"dataPath": "params_shard_58.bin",
"format": "raw-shard",
"nbytes": 32638976,
"records": [
{
"name": "model.layers.12.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.12.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.12.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.13.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039360
},
{
"name": "model.layers.13.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 17049600
},
{
"name": "model.layers.13.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 21473280
},
{
"name": "model.layers.13.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30320640
},
{
"name": "model.layers.13.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 30330880
},
{
"name": "model.layers.13.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 30345216
}
],
"md5sum": "b5d059f3f65f39a79db40ef429091666"
},
{
"dataPath": "params_shard_59.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.14.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "ed2093b10dc780f9ac64974a60fd4ca0"
},
{
"dataPath": "params_shard_60.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.14.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "d5cb087f3ca97b67bbd7a90142154521"
},
{
"dataPath": "params_shard_61.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.14.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "890f79c4f24d2a54bb86e598e44f68a5"
},
{
"dataPath": "params_shard_62.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.13.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.13.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.14.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.14.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.14.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.14.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.14.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.14.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "61237640ab6733b38f97c43364a9c156"
},
{
"dataPath": "params_shard_63.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.15.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "2c11890f47f35c8c5a6ab895da2cbd94"
},
{
"dataPath": "params_shard_64.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.15.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "309bc2d02eb4059ff81818fafd39681a"
},
{
"dataPath": "params_shard_65.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.15.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "0be40554674068aee8764267494f5a17"
},
{
"dataPath": "params_shard_66.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.14.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.14.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.15.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.15.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.15.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.15.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.15.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.15.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "4d0ba59e605c9f556c0a3d139defd34d"
},
{
"dataPath": "params_shard_67.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.16.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "3d5eb118292effd317dee88647203bad"
},
{
"dataPath": "params_shard_68.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.16.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "0f0f83ea0ae732e620d02d93d3ebae2e"
},
{
"dataPath": "params_shard_69.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.16.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "aa3b902bbf6351b5e0215da5aa035736"
},
{
"dataPath": "params_shard_70.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.15.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.15.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.16.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.16.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.16.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.16.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.16.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.16.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "4850a102e15225cdec7a134844f4349f"
},
{
"dataPath": "params_shard_71.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.17.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "5829666f9e3b93c0fd79d75e05e76b47"
},
{
"dataPath": "params_shard_72.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.17.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "a1870f6027093325c1b650effb355cd3"
},
{
"dataPath": "params_shard_73.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.17.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "76a3cb5aaa8e9a7f606809d1bdb69531"
},
{
"dataPath": "params_shard_74.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.16.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.16.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.17.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.17.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.17.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.17.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.17.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.17.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "085ce479fa5124963e7ef44d3d8692c8"
},
{
"dataPath": "params_shard_75.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.18.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "b436926f91a897d92bd301a5598ef1bd"
},
{
"dataPath": "params_shard_76.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.18.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "2edfbc680cf216aae425a396bb5d696b"
},
{
"dataPath": "params_shard_77.bin",
"format": "raw-shard",
"nbytes": 25921536,
"records": [
{
"name": "model.layers.17.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.17.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.18.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.18.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 14755840
},
{
"name": "model.layers.18.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 23603200
},
{
"name": "model.layers.18.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 23613440
},
{
"name": "model.layers.18.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 23627776
}
],
"md5sum": "fd039967d2a8f0d5a43e0ff74cbc19cf"
},
{
"dataPath": "params_shard_78.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.18.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "abcac7c4752dd589a493cf3557826a28"
},
{
"dataPath": "params_shard_79.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.19.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "c800d96c9b38b35c5ff0e3234d955614"
},
{
"dataPath": "params_shard_80.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.19.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "3b6d985d95b868cb523a77bf1e81a9a8"
},
{
"dataPath": "params_shard_81.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.19.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "e5d169d81204c13c0efc421cf4fbe2b6"
},
{
"dataPath": "params_shard_82.bin",
"format": "raw-shard",
"nbytes": 32475136,
"records": [
{
"name": "model.layers.18.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.18.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.18.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14745600
},
{
"name": "model.layers.19.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 19169280
},
{
"name": "model.layers.19.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 19179520
},
{
"name": "model.layers.19.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 23603200
},
{
"name": "model.layers.19.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 32450560
},
{
"name": "model.layers.19.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 32460800
}
],
"md5sum": "98ce84677d8f9128547e94c221236519"
},
{
"dataPath": "params_shard_83.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.20.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "3f3bde616eef73e9ed73b88f4967e481"
},
{
"dataPath": "params_shard_84.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.20.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "8f6caf8474d672446681bd7c26c8bb4e"
},
{
"dataPath": "params_shard_85.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.20.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "88a1ce37bc5f004aa20c491fc4116695"
},
{
"dataPath": "params_shard_86.bin",
"format": "raw-shard",
"nbytes": 32638976,
"records": [
{
"name": "model.layers.19.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.19.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.19.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.20.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039360
},
{
"name": "model.layers.20.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 17049600
},
{
"name": "model.layers.20.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 21473280
},
{
"name": "model.layers.20.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30320640
},
{
"name": "model.layers.20.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 30330880
},
{
"name": "model.layers.20.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 30345216
}
],
"md5sum": "3ac9677b99744bb4acb022bab6c7bc16"
},
{
"dataPath": "params_shard_87.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.21.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "6562caf90a09c52f4f7278c978fd4d5f"
},
{
"dataPath": "params_shard_88.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.21.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "14ba91369ef9c3bc0eeb871eb644e4b2"
},
{
"dataPath": "params_shard_89.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.21.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "b633d8ff47193818734fe1c95e061c81"
},
{
"dataPath": "params_shard_90.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.20.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.20.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.21.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.21.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.21.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.21.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.21.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.21.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "c6bd6ccb093eff26d5f2a18746302a7f"
},
{
"dataPath": "params_shard_91.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.22.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "13a1a218ea5bc2154cf15282a0c7f449"
},
{
"dataPath": "params_shard_92.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.22.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "5c76ba611259edf4972e1e80bb919645"
},
{
"dataPath": "params_shard_93.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.22.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "269038d9388a710d2e39313ea9bbbc81"
},
{
"dataPath": "params_shard_94.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.21.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.21.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.22.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.22.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.22.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.22.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.22.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.22.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "7b62c482a01214a0465676ade2b70327"
},
{
"dataPath": "params_shard_95.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.23.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "7649a0d2113534cf648b024361623ef7"
},
{
"dataPath": "params_shard_96.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.23.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "83ce5089894e67bdfd22200630c2dc4f"
},
{
"dataPath": "params_shard_97.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.23.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "ac6fcd50ca8c844d022d68a97a69893e"
},
{
"dataPath": "params_shard_98.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.22.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.22.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.23.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.23.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.23.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.23.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.23.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.23.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "ea8fc284f323b2b6c938dff11e43c70e"
},
{
"dataPath": "params_shard_99.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.24.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "e2203daf308da4d22140548fb1f34717"
},
{
"dataPath": "params_shard_100.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.24.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "d2753efa0768d49a3bff907955d161f4"
},
{
"dataPath": "params_shard_101.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.24.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "e444a85c1fd5104fa0a20828a30fa952"
},
{
"dataPath": "params_shard_102.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.23.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.23.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.24.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.24.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.24.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.24.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.24.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.24.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "a706e472d132561b2cc9f45426162055"
},
{
"dataPath": "params_shard_103.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.25.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "ae12e622424fdd3b0c99ae39f66500a3"
},
{
"dataPath": "params_shard_104.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.25.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "4728d22b257a852914a45afa39ff5475"
},
{
"dataPath": "params_shard_105.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.25.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "fea8f6b40725f69a37bfb72fc02ac953"
},
{
"dataPath": "params_shard_106.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.24.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.24.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.25.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.25.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.25.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.25.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.25.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.25.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "1c3f6ea01ce7b2ade06fe69be59fd6c0"
},
{
"dataPath": "params_shard_107.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.26.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "caf17274cc3eba48c92dcdf91fc936e1"
},
{
"dataPath": "params_shard_108.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.26.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "c842224f4426f87545f9660fb3df70a1"
},
{
"dataPath": "params_shard_109.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.26.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "29a98aa6f97a87de23b7c859912343e1"
},
{
"dataPath": "params_shard_110.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.25.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.25.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.26.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.26.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.26.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.26.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.26.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.26.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "ad8e91a136de776c72ed6802c1b7697c"
},
{
"dataPath": "params_shard_111.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.27.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "0cb3d108139cfa230583b3f176a870a1"
},
{
"dataPath": "params_shard_112.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.27.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "5529e49b9ae1d0187be1a790fbc5740a"
},
{
"dataPath": "params_shard_113.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.27.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "b8259b1ceebff048112180c5a87ec31d"
},
{
"dataPath": "params_shard_114.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.26.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.26.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.27.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.27.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.27.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.27.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.27.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.27.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "b9889fddb979b13a5de35145ed68e6bc"
},
{
"dataPath": "params_shard_115.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.28.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "78320e386ee72a377557a2f3e186a0fb"
},
{
"dataPath": "params_shard_116.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.28.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "bfac3caba070fa36cc8fd8ffe7fcc271"
},
{
"dataPath": "params_shard_117.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.28.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "781d064a588c11c16f24c8349495f837"
},
{
"dataPath": "params_shard_118.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.27.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.27.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.28.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.28.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.28.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.28.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.28.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.28.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "97d92e6390831dc53f1dac07d8a616ee"
},
{
"dataPath": "params_shard_119.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.29.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "8b167ebd80c637af4d53aabf45c01aa9"
},
{
"dataPath": "params_shard_120.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.29.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "7d2c446632bfdccb7369a7622e443daf"
},
{
"dataPath": "params_shard_121.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.29.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "4e3e91cdb8d126b8c76c5d0b26eab2c0"
},
{
"dataPath": "params_shard_122.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.28.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.28.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.29.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.29.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.29.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.29.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.29.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.29.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "4fc254b49ddb78cd01f536c10f2557a4"
},
{
"dataPath": "params_shard_123.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.30.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "14b0dbec993a0885c9efd67e218e7e73"
},
{
"dataPath": "params_shard_124.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.30.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "16f1a26544684003f2814fd2f8e4c479"
},
{
"dataPath": "params_shard_125.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.30.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "bc60d2cb87ee5abf3dc49cbfa596ee4e"
},
{
"dataPath": "params_shard_126.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.29.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.29.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.30.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.30.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.30.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.30.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.30.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.30.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "99f81a6555611fc636524ad3f8c6f5ba"
},
{
"dataPath": "params_shard_127.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.31.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "f307521b3ac462a01886d2a43b280cc5"
},
{
"dataPath": "params_shard_128.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.31.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "4b5e55e4190435aa39209060470168ff"
},
{
"dataPath": "params_shard_129.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.31.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "d6237a553569023c7fb2952fca0ba142"
},
{
"dataPath": "params_shard_130.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.30.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.30.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.31.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.31.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.31.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.31.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.31.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.31.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "330ba6f0790cc42ebee034d1dfbb1703"
},
{
"dataPath": "params_shard_131.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.32.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "0191a79a0ace7f80604a4fe0bc9ec8cb"
},
{
"dataPath": "params_shard_132.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.32.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "5920354e8b783fd4980fd2c410dd2ced"
},
{
"dataPath": "params_shard_133.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.32.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "c44e0892794d43b9e5787a1d71d1e0d6"
},
{
"dataPath": "params_shard_134.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.31.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.31.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.32.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.32.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.32.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.32.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.32.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.32.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "2f45e05cfbe19677f94bd9dfa9ac7263"
},
{
"dataPath": "params_shard_135.bin",
"format": "raw-shard",
"nbytes": 33130496,
"records": [
{
"name": "model.layers.32.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.32.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.33.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.33.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14755840
},
{
"name": "model.layers.33.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 14766080
},
{
"name": "model.layers.33.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 14780416
}
],
"md5sum": "7d222378bb30b81633688d12a6067bb7"
},
{
"dataPath": "params_shard_136.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.33.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "cba920ca2293f8a882687122ff4cf6ca"
},
{
"dataPath": "params_shard_137.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.33.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "af3d33837d41b2d12e46dd2120c077d1"
},
{
"dataPath": "params_shard_138.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.34.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "bcd6de59412e364f7ddce1ebc0c43f8d"
},
{
"dataPath": "params_shard_139.bin",
"format": "raw-shard",
"nbytes": 30320640,
"records": [
{
"name": "model.layers.33.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.33.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.33.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.33.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 17039360
},
{
"name": "model.layers.33.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 21463040
},
{
"name": "model.layers.34.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30310400
}
],
"md5sum": "e17dfccb9676fe1b0358e0684263001c"
},
{
"dataPath": "params_shard_140.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.34.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "6896e1b30d36b995220d9df7d377585b"
},
{
"dataPath": "params_shard_141.bin",
"format": "raw-shard",
"nbytes": 31645696,
"records": [
{
"name": "model.layers.34.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 0
},
{
"name": "model.layers.34.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 4423680
},
{
"name": "model.layers.34.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 13271040
},
{
"name": "model.layers.34.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 13281280
},
{
"name": "model.layers.34.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 13295616
}
],
"md5sum": "054d9bb2f565211b1eb723df60be8a71"
},
{
"dataPath": "params_shard_142.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.35.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "79e45e28e799fe01371e6706e100eda3"
},
{
"dataPath": "params_shard_143.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.35.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "a56af6a3f3940fbb83d4435c2f20d883"
},
{
"dataPath": "params_shard_144.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.35.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "448eca14d15ef32a1c400d2290642c0f"
},
{
"dataPath": "params_shard_145.bin",
"format": "raw-shard",
"nbytes": 32638976,
"records": [
{
"name": "model.layers.34.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.34.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.34.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.35.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039360
},
{
"name": "model.layers.35.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 17049600
},
{
"name": "model.layers.35.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 21473280
},
{
"name": "model.layers.35.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30320640
},
{
"name": "model.layers.35.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 30330880
},
{
"name": "model.layers.35.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 30345216
}
],
"md5sum": "adfdaf44486ee61377be57a999613eb2"
},
{
"dataPath": "params_shard_146.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.36.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "5bec99a0fbd860c32ef4d2b8f5c12575"
},
{
"dataPath": "params_shard_147.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.36.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "d376e6a8d828464445ff920af6691f87"
},
{
"dataPath": "params_shard_148.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.36.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "5ab055a6c91eebf3c8d48d0f57d024f7"
},
{
"dataPath": "params_shard_149.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.35.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.35.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.36.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.36.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.36.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.36.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.36.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.36.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "74bcf8ea90601d4a658c600781deb562"
},
{
"dataPath": "params_shard_150.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.37.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "920d1a1f587826b0f702849eea13fcbc"
},
{
"dataPath": "params_shard_151.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.37.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "0cad4911f71019465650b699d07cfabb"
},
{
"dataPath": "params_shard_152.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.37.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "a0f068de92fde8c59c70ca688f7fdae6"
},
{
"dataPath": "params_shard_153.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.36.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.36.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.37.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.37.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.37.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.37.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.37.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.37.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "85c1e2af540650e6f90962abee9ba332"
},
{
"dataPath": "params_shard_154.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.38.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "29a0feb54663ceac681ba7712fa8f414"
},
{
"dataPath": "params_shard_155.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.38.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "d7b2693f57e7f0a0051098619f2833d3"
},
{
"dataPath": "params_shard_156.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.38.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "6d082c7220e082bde38885d3d973e20b"
},
{
"dataPath": "params_shard_157.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.37.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.37.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.38.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.38.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.38.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.38.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.38.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.38.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "8962b611965a4a7f3ad77e75489f52a9"
},
{
"dataPath": "params_shard_158.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.39.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "0d7f7bb5edb84810c156fad475158e68"
},
{
"dataPath": "params_shard_159.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.39.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "5902c8488b6d1cbc63e142fcd475e175"
},
{
"dataPath": "params_shard_160.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.39.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "66d25bdc630cc7c7040a25370217e70a"
},
{
"dataPath": "params_shard_161.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.38.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.38.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.39.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.39.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.39.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.39.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.39.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.39.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "f58874d10ed0d1811da4c2e2d6ca0b79"
},
{
"dataPath": "params_shard_162.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.40.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "655dea053cffa3d74f1377ade3fc61c3"
},
{
"dataPath": "params_shard_163.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.40.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "d56d605169ea3a4a4b2d9ba88f3ff54f"
},
{
"dataPath": "params_shard_164.bin",
"format": "raw-shard",
"nbytes": 25921536,
"records": [
{
"name": "model.layers.39.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.39.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.40.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.40.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 14755840
},
{
"name": "model.layers.40.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 23603200
},
{
"name": "model.layers.40.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 23613440
},
{
"name": "model.layers.40.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 23627776
}
],
"md5sum": "a3f920aee59f8a9b377f7889ac83d5eb"
},
{
"dataPath": "params_shard_165.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.40.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "3a08c65fb76a3c82c52ddf189effa8b0"
},
{
"dataPath": "params_shard_166.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.41.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "e0ab44f4d7a2d6f7ecff2536ad47d945"
},
{
"dataPath": "params_shard_167.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.41.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "fdc591b02c23c0a7d6ea6775ad7b0cc0"
},
{
"dataPath": "params_shard_168.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.41.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "411fc16dc17dfb268cce02e915027c10"
},
{
"dataPath": "params_shard_169.bin",
"format": "raw-shard",
"nbytes": 32475136,
"records": [
{
"name": "model.layers.40.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.40.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.40.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14745600
},
{
"name": "model.layers.41.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 19169280
},
{
"name": "model.layers.41.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 19179520
},
{
"name": "model.layers.41.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 23603200
},
{
"name": "model.layers.41.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 32450560
},
{
"name": "model.layers.41.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 32460800
}
],
"md5sum": "e9eaaec5f56c3b62630d57a7915c053a"
},
{
"dataPath": "params_shard_170.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.42.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "5b40b287953579f177b0433c205123a9"
},
{
"dataPath": "params_shard_171.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.42.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "0100394dc654cbb1756ce3e635f10fea"
},
{
"dataPath": "params_shard_172.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.42.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "8e584f27e94e6d0efddf140d03c45957"
},
{
"dataPath": "params_shard_173.bin",
"format": "raw-shard",
"nbytes": 32638976,
"records": [
{
"name": "model.layers.41.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 0
},
{
"name": "model.layers.41.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 2293760
},
{
"name": "model.layers.41.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 15400960
},
{
"name": "model.layers.42.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 17039360
},
{
"name": "model.layers.42.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 17049600
},
{
"name": "model.layers.42.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 21473280
},
{
"name": "model.layers.42.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 30320640
},
{
"name": "model.layers.42.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 30330880
},
{
"name": "model.layers.42.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 30345216
}
],
"md5sum": "24018a8f264a06416302c8d1927d4860"
},
{
"dataPath": "params_shard_174.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.43.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "933193ae209bd0f689b033686dc453a0"
},
{
"dataPath": "params_shard_175.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.43.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "4c9c27abee39792d1175ca7a963994f0"
},
{
"dataPath": "params_shard_176.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.43.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "932f8177e5f2bb0747b5c4516b5f067b"
},
{
"dataPath": "params_shard_177.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.42.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.42.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.43.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.43.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.43.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.43.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.43.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.43.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "7488981ef267a7b008c1e9ba79792017"
},
{
"dataPath": "params_shard_178.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.44.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "2dd16300d2eb099f28522d6bfa3c886e"
},
{
"dataPath": "params_shard_179.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.44.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "17ec73693e78331d9c2ca3fcb136d9a1"
},
{
"dataPath": "params_shard_180.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.44.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "8b89289fb66197b4604ceafcd48cd7ec"
},
{
"dataPath": "params_shard_181.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.43.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.43.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.44.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.44.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.44.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.44.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.44.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.44.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "70b543ed9e83afe9cc60b8247fb2786d"
},
{
"dataPath": "params_shard_182.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.45.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "ced520a8b8ff21cad2077a6d2a35ff13"
},
{
"dataPath": "params_shard_183.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.45.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "7663d8a11b5dc2c91d435c861178be80"
},
{
"dataPath": "params_shard_184.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.45.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "6cc9b6d67bd97d33533868666fe55c81"
},
{
"dataPath": "params_shard_185.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.44.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.44.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.45.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.45.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.45.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.45.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.45.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.45.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "96f35fee8272fa13b5ab1c96b026789c"
},
{
"dataPath": "params_shard_186.bin",
"format": "raw-shard",
"nbytes": 35389440,
"records": [
{
"name": "model.layers.46.mlp.down_proj.q_weight",
"shape": [
5120,
1728
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 35389440,
"byteOffset": 0
}
],
"md5sum": "60dade9e66a3f7f8bcc475abb855d9e7"
},
{
"dataPath": "params_shard_187.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.46.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "aa6f0ce64c44c9866b68544aae9d53a5"
},
{
"dataPath": "params_shard_188.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.46.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "f6249a1ecdda8cfbd49df4a7365e58e9"
},
{
"dataPath": "params_shard_189.bin",
"format": "raw-shard",
"nbytes": 30345216,
"records": [
{
"name": "model.layers.45.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.45.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.46.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.46.mlp.down_proj.q_scale",
"shape": [
5120,
432
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4423680,
"byteOffset": 14755840
},
{
"name": "model.layers.46.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 19179520
},
{
"name": "model.layers.46.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 28026880
},
{
"name": "model.layers.46.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 28037120
},
{
"name": "model.layers.46.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 28051456
}
],
"md5sum": "61952474541f54d3cbc42073a536e39c"
},
{
"dataPath": "params_shard_190.bin",
"format": "raw-shard",
"nbytes": 70778880,
"records": [
{
"name": "model.layers.47.mlp.gate_up_proj.q_weight",
"shape": [
27648,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 70778880,
"byteOffset": 0
}
],
"md5sum": "ae4ac7d2dabd7d9a92482dfcd63cd6a0"
},
{
"dataPath": "params_shard_191.bin",
"format": "raw-shard",
"nbytes": 18350080,
"records": [
{
"name": "model.layers.47.self_attn.c_attn.q_weight",
"shape": [
7168,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 18350080,
"byteOffset": 0
}
],
"md5sum": "4d39a5db4a36e68984a52d36f8cd9147"
},
{
"dataPath": "params_shard_192.bin",
"format": "raw-shard",
"nbytes": 25921536,
"records": [
{
"name": "model.layers.46.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.46.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "model.layers.47.input_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 14745600
},
{
"name": "model.layers.47.mlp.gate_up_proj.q_scale",
"shape": [
27648,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 8847360,
"byteOffset": 14755840
},
{
"name": "model.layers.47.post_attention_layernorm.weight",
"shape": [
5120
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 10240,
"byteOffset": 23603200
},
{
"name": "model.layers.47.self_attn.c_attn.bias",
"shape": [
7168
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 14336,
"byteOffset": 23613440
},
{
"name": "model.layers.47.self_attn.c_attn.q_scale",
"shape": [
7168,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2293760,
"byteOffset": 23627776
}
],
"md5sum": "7b713782d508fddccb1767bc1878c9ad"
},
{
"dataPath": "params_shard_193.bin",
"format": "raw-shard",
"nbytes": 14745600,
"records": [
{
"name": "model.layers.47.self_attn.o_proj.q_weight",
"shape": [
5120,
640
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "model.layers.47.self_attn.o_proj.q_scale",
"shape": [
5120,
160
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1638400,
"byteOffset": 13107200
}
],
"md5sum": "8910d8d3e1c16f69f46a58ca01c99bca"
}
]
}