|
{ |
|
"metadata": { |
|
"ParamSize": 325, |
|
"ParamBytes": 2149644288.0, |
|
"BitsPerParam": 4.500600961055312 |
|
}, |
|
"records": [ |
|
{ |
|
"dataPath": "params_shard_0.bin", |
|
"format": "raw-shard", |
|
"nbytes": 49250304, |
|
"records": [ |
|
{ |
|
"name": "lm_head.q_weight", |
|
"shape": [ |
|
384, |
|
32064 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 49250304, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "2059e8510f2bb7c5aaa04f7a34635e7e" |
|
}, |
|
{ |
|
"dataPath": "params_shard_1.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.21.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "81c093a6405b3449faf1ea3d4cbfdfca" |
|
}, |
|
{ |
|
"dataPath": "params_shard_2.bin", |
|
"format": "raw-shard", |
|
"nbytes": 23470080, |
|
"records": [ |
|
{ |
|
"name": "lm_head.q_scale", |
|
"shape": [ |
|
96, |
|
32064 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6156288, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.21.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 6156288 |
|
}, |
|
{ |
|
"name": "transformer.h.21.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 6162432 |
|
}, |
|
{ |
|
"name": "transformer.h.21.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 18745344 |
|
}, |
|
{ |
|
"name": "transformer.h.21.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 20318208 |
|
}, |
|
{ |
|
"name": "transformer.h.21.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 23463936 |
|
} |
|
], |
|
"md5sum": "f2a248970ed31f9c7c619a54bd543953" |
|
}, |
|
{ |
|
"dataPath": "params_shard_3.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.22.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "54f9ef0fc00897b5bcbd4e3b912f959f" |
|
}, |
|
{ |
|
"dataPath": "params_shard_4.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.21.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.21.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.22.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.22.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.22.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.22.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.22.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "3e50e5db4e4d9dadae97b307fcf73d70" |
|
}, |
|
{ |
|
"dataPath": "params_shard_5.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.22.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.22.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.22.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.22.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.23.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "f1e10e96ac5f24ac0afe3a0d5a8600fa" |
|
}, |
|
{ |
|
"dataPath": "params_shard_6.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.23.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "f4655284a2e3a2c7de3632eb7b803939" |
|
}, |
|
{ |
|
"dataPath": "params_shard_7.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.23.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.23.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.23.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.23.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.23.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.23.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "0c3da63ed54fff25eea12ed93c70eb9e" |
|
}, |
|
{ |
|
"dataPath": "params_shard_8.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.24.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "3c29c491a70bb69ddf99e4085362ff5b" |
|
}, |
|
{ |
|
"dataPath": "params_shard_9.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.23.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.23.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.24.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.24.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.24.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.24.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.24.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "954e9e57eebc79033bbe7df64cd7be31" |
|
}, |
|
{ |
|
"dataPath": "params_shard_10.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.24.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.24.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.24.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.24.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.25.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "52529802bee1a8a943e0b7257c4202b6" |
|
}, |
|
{ |
|
"dataPath": "params_shard_11.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.25.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "5ddd842dd894fd65540b5e747a2bbd30" |
|
}, |
|
{ |
|
"dataPath": "params_shard_12.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.25.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.25.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.25.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.25.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.25.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.25.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "8b99da6b83d36988375a2518e2c631a1" |
|
}, |
|
{ |
|
"dataPath": "params_shard_13.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.26.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "d5f13eed3cac63d2623ee9a472677d30" |
|
}, |
|
{ |
|
"dataPath": "params_shard_14.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.25.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.25.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.26.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.26.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.26.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.26.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.26.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "5d7205d5243285fa84106f87887b68fa" |
|
}, |
|
{ |
|
"dataPath": "params_shard_15.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.26.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.26.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.26.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.26.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.27.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "5787f02404c876752e0acab5e116234c" |
|
}, |
|
{ |
|
"dataPath": "params_shard_16.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.27.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "26162aabebcf2c9e84d5ea9d01f68a89" |
|
}, |
|
{ |
|
"dataPath": "params_shard_17.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.27.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.27.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.27.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.27.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.27.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.27.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "77f66d6a0a219eb0203870b4f9094b3c" |
|
}, |
|
{ |
|
"dataPath": "params_shard_18.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.28.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "18e2955c8f884d2ccfdf4a537ce9982a" |
|
}, |
|
{ |
|
"dataPath": "params_shard_19.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.27.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.27.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.28.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.28.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.28.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.28.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.28.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "ae05bc4ee20105f8a6e67fae9997360e" |
|
}, |
|
{ |
|
"dataPath": "params_shard_20.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.28.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.28.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.28.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.28.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.29.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "df25ebb89340f015b256c773978d109b" |
|
}, |
|
{ |
|
"dataPath": "params_shard_21.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.29.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "9920b0466ba3854bd6b40f593118f8d5" |
|
}, |
|
{ |
|
"dataPath": "params_shard_22.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.29.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.29.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.29.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.29.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.29.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.29.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "77eb8127c89fc163eaa24bc87c5f5f4b" |
|
}, |
|
{ |
|
"dataPath": "params_shard_23.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.30.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "49a460541ff5a7458b062fb18e22bfc2" |
|
}, |
|
{ |
|
"dataPath": "params_shard_24.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.29.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.29.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.30.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.30.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.30.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.30.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.30.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "d02fdebb2f9d690a80b000924624f8d5" |
|
}, |
|
{ |
|
"dataPath": "params_shard_25.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.30.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.30.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.30.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.30.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.31.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "878e23d8fa2464becbd7079bfe4c2b8c" |
|
}, |
|
{ |
|
"dataPath": "params_shard_26.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.31.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "14091134e5c6d814ea9961c6596a2cb7" |
|
}, |
|
{ |
|
"dataPath": "params_shard_27.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.31.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.31.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.31.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.31.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.31.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.31.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "ed0e2c3abdf85f35c44cd6341de8f3b2" |
|
}, |
|
{ |
|
"dataPath": "params_shard_28.bin", |
|
"format": "raw-shard", |
|
"nbytes": 49250304, |
|
"records": [ |
|
{ |
|
"name": "transformer.embd.q_weight", |
|
"shape": [ |
|
32064, |
|
384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 49250304, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "3f9a3cb77fdaedfd62e9720fc14e716e" |
|
}, |
|
{ |
|
"dataPath": "params_shard_29.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22093824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.31.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.31.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.norm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.embd.q_scale", |
|
"shape": [ |
|
32064, |
|
96 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6156288, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.0.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 22087680 |
|
} |
|
], |
|
"md5sum": "906df66a7a47401be015e733e9aeb10d" |
|
}, |
|
{ |
|
"dataPath": "params_shard_30.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.0.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "8d083ebbc92427a48d1bd76b1ab44669" |
|
}, |
|
{ |
|
"dataPath": "params_shard_31.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.0.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.0.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.0.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.0.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.0.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.0.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "0c2d54874ac0f0d475292316f7c082af" |
|
}, |
|
{ |
|
"dataPath": "params_shard_32.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.1.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "ed76edd70787f5a9eddb86c5751a3baf" |
|
}, |
|
{ |
|
"dataPath": "params_shard_33.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.0.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.0.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.1.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.1.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.1.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.1.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.1.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "24cf5c865f88ee5ca7d71ef25a87c455" |
|
}, |
|
{ |
|
"dataPath": "params_shard_34.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.1.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.1.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.1.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.1.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.10.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "56dcf967c263478ac2f318e8c2555089" |
|
}, |
|
{ |
|
"dataPath": "params_shard_35.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.10.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "86c9c7810e1286978ae229ba21403479" |
|
}, |
|
{ |
|
"dataPath": "params_shard_36.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.10.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.10.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.10.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.10.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.10.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.10.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "944f57f46be953461f246c5363c8e2b9" |
|
}, |
|
{ |
|
"dataPath": "params_shard_37.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.11.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "cc9d04ff22069d23ec8b6b287dab9f34" |
|
}, |
|
{ |
|
"dataPath": "params_shard_38.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.10.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.10.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.11.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.11.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.11.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.11.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.11.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "c39f34a0dd42f7450964d934a713bf7d" |
|
}, |
|
{ |
|
"dataPath": "params_shard_39.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.11.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.11.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.11.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.11.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.12.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "02b86e6c64bf07620f35c8acc92cfc96" |
|
}, |
|
{ |
|
"dataPath": "params_shard_40.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.12.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "59e368d72821d04e59fd252d917ebf2e" |
|
}, |
|
{ |
|
"dataPath": "params_shard_41.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.12.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.12.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.12.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.12.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.12.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.12.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "86a4c47aecd5a4710f636efa0c645329" |
|
}, |
|
{ |
|
"dataPath": "params_shard_42.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.13.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "35d2a3d4efc061db1af7b497332f8c81" |
|
}, |
|
{ |
|
"dataPath": "params_shard_43.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.12.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.12.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.13.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.13.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.13.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.13.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.13.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "2a6bdecb0d0bc179b5ce5ff88f89fb5c" |
|
}, |
|
{ |
|
"dataPath": "params_shard_44.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.13.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.13.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.13.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.13.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.14.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "4a54866e97bf9d1d45ae23917febae5c" |
|
}, |
|
{ |
|
"dataPath": "params_shard_45.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.14.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "cfe8d7de206f1e8eed0ff4d0cc6e658a" |
|
}, |
|
{ |
|
"dataPath": "params_shard_46.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.14.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.14.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.14.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.14.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.14.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.14.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "d449f204983e57a997fbe60d94ff1e3d" |
|
}, |
|
{ |
|
"dataPath": "params_shard_47.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.15.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "702c7ac5b8e2e506101718c984af0796" |
|
}, |
|
{ |
|
"dataPath": "params_shard_48.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.14.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.14.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.15.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.15.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.15.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.15.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.15.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "80153fda145160a17d1003bb245f791e" |
|
}, |
|
{ |
|
"dataPath": "params_shard_49.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.15.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.15.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.15.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.15.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.16.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "77a25b3156f472ef13c46efff7e187fb" |
|
}, |
|
{ |
|
"dataPath": "params_shard_50.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.16.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "678545c2d04f1c3fd74bb0eba44c5a31" |
|
}, |
|
{ |
|
"dataPath": "params_shard_51.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.16.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.16.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.16.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.16.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.16.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.16.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "77136c98849ee667df7bdaf117c84005" |
|
}, |
|
{ |
|
"dataPath": "params_shard_52.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.17.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "1f62db8cfa0c12b8e32f7aebb6694e91" |
|
}, |
|
{ |
|
"dataPath": "params_shard_53.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.16.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.16.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.17.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.17.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.17.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.17.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.17.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "be8594200b3d5983ef429b2fd6496d14" |
|
}, |
|
{ |
|
"dataPath": "params_shard_54.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.17.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.17.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.17.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.17.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.18.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "36ecd1d76a0f4673841e66f4c3eefe35" |
|
}, |
|
{ |
|
"dataPath": "params_shard_55.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.18.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "8b51615e5e3df98345edf4d18793790c" |
|
}, |
|
{ |
|
"dataPath": "params_shard_56.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.18.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.18.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.18.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.18.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.18.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.18.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "103ac3616502b89cb0ddf574bd3b8155" |
|
}, |
|
{ |
|
"dataPath": "params_shard_57.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.19.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "f4fff3486a085ac3a2304623f6538d7b" |
|
}, |
|
{ |
|
"dataPath": "params_shard_58.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.18.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.18.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.19.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.19.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.19.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.19.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.19.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "738612fc811b24ff2eec5295f4321c21" |
|
}, |
|
{ |
|
"dataPath": "params_shard_59.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.19.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.19.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.19.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.19.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.2.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "e713cc90fffcf73a7e8140d8ed257dcd" |
|
}, |
|
{ |
|
"dataPath": "params_shard_60.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.2.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "43552c4cdf6f150def58ae644983b9e2" |
|
}, |
|
{ |
|
"dataPath": "params_shard_61.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.2.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.2.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.2.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.2.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.2.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.2.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "af622c19806902d8f0c304a9db04c466" |
|
}, |
|
{ |
|
"dataPath": "params_shard_62.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.20.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "a7a7a651454ebaa5a00f119c3dda2ac8" |
|
}, |
|
{ |
|
"dataPath": "params_shard_63.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.2.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.2.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.20.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.20.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.20.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.20.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.20.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "cc120f3b34a97815b72433e3d27ee550" |
|
}, |
|
{ |
|
"dataPath": "params_shard_64.bin", |
|
"format": "raw-shard", |
|
"nbytes": 26548224, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.20.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.20.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.20.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.20.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.21.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 21233664 |
|
}, |
|
{ |
|
"name": "transformer.h.21.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 25952256 |
|
}, |
|
{ |
|
"name": "transformer.h.3.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 26542080 |
|
} |
|
], |
|
"md5sum": "531a540d4864984186aa77683a2fde1c" |
|
}, |
|
{ |
|
"dataPath": "params_shard_65.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.3.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "f919604b0e1e5b57d16395ddd236aa7c" |
|
}, |
|
{ |
|
"dataPath": "params_shard_66.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.3.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.3.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.3.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.3.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.3.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.3.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "ed238511ae92cea886d4bb5beb791d98" |
|
}, |
|
{ |
|
"dataPath": "params_shard_67.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.4.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "1714eb755b54f26437097d4345327ae6" |
|
}, |
|
{ |
|
"dataPath": "params_shard_68.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.3.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.3.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.4.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.4.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.4.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.4.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.4.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "536e10621663a5fbdebc7c05433978e2" |
|
}, |
|
{ |
|
"dataPath": "params_shard_69.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.4.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.4.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.4.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.4.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.5.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "514f3639dec43f41c1aa3c265c499da2" |
|
}, |
|
{ |
|
"dataPath": "params_shard_70.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.5.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "71ee17f7dc2beee391891de2250c7247" |
|
}, |
|
{ |
|
"dataPath": "params_shard_71.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.5.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.5.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.5.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.5.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.5.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.5.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "84cf6e8a1795a4bb7dd51edc17e66f3d" |
|
}, |
|
{ |
|
"dataPath": "params_shard_72.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.6.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "0659e9db0255594be15d875215fab030" |
|
}, |
|
{ |
|
"dataPath": "params_shard_73.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.5.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.5.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.6.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.6.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.6.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.6.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.6.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "0723f928b545286ef4f426de86b187b4" |
|
}, |
|
{ |
|
"dataPath": "params_shard_74.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.6.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.6.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.6.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.6.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.7.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "453a1316ad9c724cc49798bfc97fb743" |
|
}, |
|
{ |
|
"dataPath": "params_shard_75.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.7.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "326f2dcc421cdc6a37c09a0624b9c865" |
|
}, |
|
{ |
|
"dataPath": "params_shard_76.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.7.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.7.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.7.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.7.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.7.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.7.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "e85c95119a6f53114b6a0268c311e370" |
|
}, |
|
{ |
|
"dataPath": "params_shard_77.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.8.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "11619ae7ca09276d9099cdcb14e75fa9" |
|
}, |
|
{ |
|
"dataPath": "params_shard_78.bin", |
|
"format": "raw-shard", |
|
"nbytes": 33239040, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.7.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.7.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.8.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 15925248 |
|
}, |
|
{ |
|
"name": "transformer.h.8.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 15931392 |
|
}, |
|
{ |
|
"name": "transformer.h.8.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 28514304 |
|
}, |
|
{ |
|
"name": "transformer.h.8.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 30087168 |
|
}, |
|
{ |
|
"name": "transformer.h.8.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 33232896 |
|
} |
|
], |
|
"md5sum": "a90e233fbb49d3647eb38c57ab318eea" |
|
}, |
|
{ |
|
"dataPath": "params_shard_79.bin", |
|
"format": "raw-shard", |
|
"nbytes": 21239808, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.8.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.8.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 4718592 |
|
}, |
|
{ |
|
"name": "transformer.h.8.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 5308416 |
|
}, |
|
{ |
|
"name": "transformer.h.8.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 19464192 |
|
}, |
|
{ |
|
"name": "transformer.h.9.ln.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 21233664 |
|
} |
|
], |
|
"md5sum": "e296e25625cc36cd2fbcf2ac88d4ca34" |
|
}, |
|
{ |
|
"dataPath": "params_shard_80.bin", |
|
"format": "raw-shard", |
|
"nbytes": 25165824, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.9.mlp.gate_up_proj.q_weight", |
|
"shape": [ |
|
384, |
|
16384 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 25165824, |
|
"byteOffset": 0 |
|
} |
|
], |
|
"md5sum": "7844854c94a298b75a2c77cd7c681ce9" |
|
}, |
|
{ |
|
"dataPath": "params_shard_81.bin", |
|
"format": "raw-shard", |
|
"nbytes": 22616064, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.9.mlp.down_proj.q_weight", |
|
"shape": [ |
|
1024, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 12582912, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.9.mlp.down_proj.q_scale", |
|
"shape": [ |
|
256, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1572864, |
|
"byteOffset": 12582912 |
|
}, |
|
{ |
|
"name": "transformer.h.9.mlp.gate_up_proj.q_scale", |
|
"shape": [ |
|
96, |
|
16384 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 3145728, |
|
"byteOffset": 14155776 |
|
}, |
|
{ |
|
"name": "transformer.h.9.post_attention_layernorm.weight", |
|
"shape": [ |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 6144, |
|
"byteOffset": 17301504 |
|
}, |
|
{ |
|
"name": "transformer.h.9.mixer.out_proj.q_weight", |
|
"shape": [ |
|
384, |
|
3072 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 4718592, |
|
"byteOffset": 17307648 |
|
}, |
|
{ |
|
"name": "transformer.h.9.mixer.out_proj.q_scale", |
|
"shape": [ |
|
96, |
|
3072 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 589824, |
|
"byteOffset": 22026240 |
|
} |
|
], |
|
"md5sum": "98ce3f01a804cfc81e865cadf0ac2e3b" |
|
}, |
|
{ |
|
"dataPath": "params_shard_82.bin", |
|
"format": "raw-shard", |
|
"nbytes": 15925248, |
|
"records": [ |
|
{ |
|
"name": "transformer.h.9.mixer.qkv_proj.q_weight", |
|
"shape": [ |
|
384, |
|
9216 |
|
], |
|
"dtype": "uint32", |
|
"format": "f32-to-bf16", |
|
"nbytes": 14155776, |
|
"byteOffset": 0 |
|
}, |
|
{ |
|
"name": "transformer.h.9.mixer.qkv_proj.q_scale", |
|
"shape": [ |
|
96, |
|
9216 |
|
], |
|
"dtype": "float16", |
|
"format": "f32-to-bf16", |
|
"nbytes": 1769472, |
|
"byteOffset": 14155776 |
|
} |
|
], |
|
"md5sum": "4b17e3e6183aea46a66906af7a9ed176" |
|
} |
|
] |
|
} |