riczhou's picture
Upload folder using huggingface_hub
98aa6b0 verified
{
"metadata": {
"ParamSize": 198,
"ParamBytes": 3087428608.0,
"BitsPerParam": 16.0
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 466747392,
"records": [
{
"name": "model.embed_tokens.weight",
"shape": [
151936,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 466747392,
"byteOffset": 0
}
],
"md5sum": "dc5478854be743693cef7917d1e7ce4c"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.0.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "e067930d718e625798bc215501ef772e"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 27535360,
"records": [
{
"name": "model.layers.0.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 0
},
{
"name": "model.layers.0.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 3072
},
{
"name": "model.layers.0.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 27528192
},
{
"name": "model.layers.0.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 27531264
}
],
"md5sum": "6e1ea98a9094cc0eb683d5027d453b18"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.1.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "ddba98c52941cd30517222e7a1e744ec"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.1.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "b43ee04337f01512b2699484f53889e2"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.10.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "6f468176df511a1de802eb813dab6b80"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.10.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "bf7baca2f207dbaa9b6a2e9bc06fae28"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.11.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "b22c5b004d6360c776f1050766a39018"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.11.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "2ab3cc8300727c2e1f7da4c5e08e9b0d"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 33060864,
"records": [
{
"name": "model.layers.0.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 0
},
{
"name": "model.layers.0.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 6291456
},
{
"name": "model.layers.1.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11010048
},
{
"name": "model.layers.1.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11013120
},
{
"name": "model.layers.1.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 11016192
},
{
"name": "model.layers.1.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 11020288
},
{
"name": "model.layers.1.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 17311744
},
{
"name": "model.layers.10.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22030336
},
{
"name": "model.layers.10.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22033408
},
{
"name": "model.layers.10.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 22036480
},
{
"name": "model.layers.10.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 22040576
},
{
"name": "model.layers.10.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 28332032
},
{
"name": "model.layers.11.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33050624
},
{
"name": "model.layers.11.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33053696
},
{
"name": "model.layers.11.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 33056768
}
],
"md5sum": "3c2d0b091ca815935b4e767fc456152a"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.12.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "8640e2a772e5f505cff9ef4d1278b993"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.12.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "1a13ad3c671d922d98496eccd823b010"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.13.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "04b10eb2f8d045366d4c14b82c37d12d"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.13.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "6350144d0f2d5da6cc69725ace1f47c4"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.14.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "78cc8750968e5324e5c549bfe82efa7f"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.14.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "f8c5b627e8354e0cafa8f690c88b0d77"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 33060864,
"records": [
{
"name": "model.layers.11.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 0
},
{
"name": "model.layers.11.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 6291456
},
{
"name": "model.layers.12.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11010048
},
{
"name": "model.layers.12.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11013120
},
{
"name": "model.layers.12.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 11016192
},
{
"name": "model.layers.12.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 11020288
},
{
"name": "model.layers.12.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 17311744
},
{
"name": "model.layers.13.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22030336
},
{
"name": "model.layers.13.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22033408
},
{
"name": "model.layers.13.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 22036480
},
{
"name": "model.layers.13.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 22040576
},
{
"name": "model.layers.13.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 28332032
},
{
"name": "model.layers.14.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33050624
},
{
"name": "model.layers.14.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33053696
},
{
"name": "model.layers.14.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 33056768
}
],
"md5sum": "df9950bdd911731e3dd85ddda77fd582"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.15.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "7aa26bb0aeed376ea1898ad7cafaed13"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.15.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "e47ac924fd3d52b09d7777158796354a"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.16.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "5b7bc689b7dfec86ef5500b58aa9a350"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.16.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "3ac080a9e359ed385700a8b5d6bfaff1"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.17.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "4e2169393a9a12174945b636f734f32c"
},
{
"dataPath": "params_shard_22.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.17.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "8b6ce185e0fed51ec8c098d879feb573"
},
{
"dataPath": "params_shard_23.bin",
"format": "raw-shard",
"nbytes": 33060864,
"records": [
{
"name": "model.layers.14.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 0
},
{
"name": "model.layers.14.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 6291456
},
{
"name": "model.layers.15.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11010048
},
{
"name": "model.layers.15.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11013120
},
{
"name": "model.layers.15.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 11016192
},
{
"name": "model.layers.15.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 11020288
},
{
"name": "model.layers.15.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 17311744
},
{
"name": "model.layers.16.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22030336
},
{
"name": "model.layers.16.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22033408
},
{
"name": "model.layers.16.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 22036480
},
{
"name": "model.layers.16.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 22040576
},
{
"name": "model.layers.16.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 28332032
},
{
"name": "model.layers.17.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33050624
},
{
"name": "model.layers.17.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33053696
},
{
"name": "model.layers.17.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 33056768
}
],
"md5sum": "7c2462692938c376178ad1c00edab3ba"
},
{
"dataPath": "params_shard_24.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.18.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "e7ba6a56026eaab113b222ed28710af7"
},
{
"dataPath": "params_shard_25.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.18.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "58523fd8e8ea53ff4120e37738bbb08f"
},
{
"dataPath": "params_shard_26.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.19.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "d3bacbc04c8bb0d0a1c592196f2e5923"
},
{
"dataPath": "params_shard_27.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.19.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "906fda15b65c863eaa42e0d018287cd2"
},
{
"dataPath": "params_shard_28.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.2.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "5b444ecd6e9f163c2dc84e01e43e30b0"
},
{
"dataPath": "params_shard_29.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.2.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "5d14245af8728a641729575cb062050a"
},
{
"dataPath": "params_shard_30.bin",
"format": "raw-shard",
"nbytes": 33060864,
"records": [
{
"name": "model.layers.17.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 0
},
{
"name": "model.layers.17.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 6291456
},
{
"name": "model.layers.18.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11010048
},
{
"name": "model.layers.18.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11013120
},
{
"name": "model.layers.18.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 11016192
},
{
"name": "model.layers.18.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 11020288
},
{
"name": "model.layers.18.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 17311744
},
{
"name": "model.layers.19.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22030336
},
{
"name": "model.layers.19.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22033408
},
{
"name": "model.layers.19.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 22036480
},
{
"name": "model.layers.19.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 22040576
},
{
"name": "model.layers.19.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 28332032
},
{
"name": "model.layers.2.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33050624
},
{
"name": "model.layers.2.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33053696
},
{
"name": "model.layers.2.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 33056768
}
],
"md5sum": "047565e1ead6ed8171acb0bb05dfe4f6"
},
{
"dataPath": "params_shard_31.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.20.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "e1920edf5a425577488f56494f8c6799"
},
{
"dataPath": "params_shard_32.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.20.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "61841b982921f56767fd49c20571f051"
},
{
"dataPath": "params_shard_33.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.21.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "93a633e61cada747ea483b5ada3e509c"
},
{
"dataPath": "params_shard_34.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.21.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "b69fb9fd8484d9f0574c48361a965cc5"
},
{
"dataPath": "params_shard_35.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.22.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "e2b90bd426da7a86cb80224a597ace2b"
},
{
"dataPath": "params_shard_36.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.22.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "fcaefdca80fe0773e56358c090a6ceb9"
},
{
"dataPath": "params_shard_37.bin",
"format": "raw-shard",
"nbytes": 33060864,
"records": [
{
"name": "model.layers.2.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 0
},
{
"name": "model.layers.2.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 6291456
},
{
"name": "model.layers.20.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11010048
},
{
"name": "model.layers.20.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11013120
},
{
"name": "model.layers.20.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 11016192
},
{
"name": "model.layers.20.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 11020288
},
{
"name": "model.layers.20.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 17311744
},
{
"name": "model.layers.21.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22030336
},
{
"name": "model.layers.21.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22033408
},
{
"name": "model.layers.21.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 22036480
},
{
"name": "model.layers.21.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 22040576
},
{
"name": "model.layers.21.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 28332032
},
{
"name": "model.layers.22.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33050624
},
{
"name": "model.layers.22.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33053696
},
{
"name": "model.layers.22.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 33056768
}
],
"md5sum": "2a96d0a5c817151eec586fceab0ab0b8"
},
{
"dataPath": "params_shard_38.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.23.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "f52e06df9574515f4877cb789944514c"
},
{
"dataPath": "params_shard_39.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.23.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "39f9ee9b27a38ea757802010872449ed"
},
{
"dataPath": "params_shard_40.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.24.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "6c7cce5cffb798ed35ae4d08df743fe0"
},
{
"dataPath": "params_shard_41.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.24.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "c7c785f203286280e3068eda1226b619"
},
{
"dataPath": "params_shard_42.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.25.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "9aa68fccf3110a34dd87bd6fa8cff3ef"
},
{
"dataPath": "params_shard_43.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.25.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "b775b1981eb39c67f4501dee796d4d82"
},
{
"dataPath": "params_shard_44.bin",
"format": "raw-shard",
"nbytes": 33060864,
"records": [
{
"name": "model.layers.22.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 0
},
{
"name": "model.layers.22.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 6291456
},
{
"name": "model.layers.23.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11010048
},
{
"name": "model.layers.23.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11013120
},
{
"name": "model.layers.23.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 11016192
},
{
"name": "model.layers.23.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 11020288
},
{
"name": "model.layers.23.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 17311744
},
{
"name": "model.layers.24.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22030336
},
{
"name": "model.layers.24.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22033408
},
{
"name": "model.layers.24.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 22036480
},
{
"name": "model.layers.24.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 22040576
},
{
"name": "model.layers.24.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 28332032
},
{
"name": "model.layers.25.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33050624
},
{
"name": "model.layers.25.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33053696
},
{
"name": "model.layers.25.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 33056768
}
],
"md5sum": "2a42cdaadb418cc8b34daf6a1b743ca8"
},
{
"dataPath": "params_shard_45.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.26.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "dde0bedc0f713bf40e8ef1006a59b027"
},
{
"dataPath": "params_shard_46.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.26.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "b1a5630fceb2992d970f390df8e593e5"
},
{
"dataPath": "params_shard_47.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.27.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "ca2b36e5409ba1d0e8fc763531eec46c"
},
{
"dataPath": "params_shard_48.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.27.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "4d2db27561b873e6f523a1ac1fc431a7"
},
{
"dataPath": "params_shard_49.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.3.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "0fb7691571e7c645aa3a5f764664b64e"
},
{
"dataPath": "params_shard_50.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.3.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "240e2893ac2a56bf4d20d112d0cdfa03"
},
{
"dataPath": "params_shard_51.bin",
"format": "raw-shard",
"nbytes": 33060864,
"records": [
{
"name": "model.layers.25.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 0
},
{
"name": "model.layers.25.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 6291456
},
{
"name": "model.layers.26.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11010048
},
{
"name": "model.layers.26.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11013120
},
{
"name": "model.layers.26.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 11016192
},
{
"name": "model.layers.26.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 11020288
},
{
"name": "model.layers.26.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 17311744
},
{
"name": "model.layers.27.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22030336
},
{
"name": "model.layers.27.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22033408
},
{
"name": "model.layers.27.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 22036480
},
{
"name": "model.layers.27.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 22040576
},
{
"name": "model.layers.27.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 28332032
},
{
"name": "model.layers.3.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33050624
},
{
"name": "model.layers.3.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33053696
},
{
"name": "model.layers.3.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 33056768
}
],
"md5sum": "084acc443ccb211387199e34607196a0"
},
{
"dataPath": "params_shard_52.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.4.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "2599506f5f40fc8e24cd03cdab25d9ec"
},
{
"dataPath": "params_shard_53.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.4.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "377ddf1e6ef01df76df6f45b47dd7159"
},
{
"dataPath": "params_shard_54.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.5.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "00f283813acc0f97f57fc4a8b3186310"
},
{
"dataPath": "params_shard_55.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "baa28a19fd8d159da7395b7428563ff9"
},
{
"dataPath": "params_shard_56.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.6.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "156da00f0de6936cf3c72aa02ec72ec3"
},
{
"dataPath": "params_shard_57.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.6.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "c42214ad3ca08e10103412dc630f3e9c"
},
{
"dataPath": "params_shard_58.bin",
"format": "raw-shard",
"nbytes": 33060864,
"records": [
{
"name": "model.layers.3.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 0
},
{
"name": "model.layers.3.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 6291456
},
{
"name": "model.layers.4.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11010048
},
{
"name": "model.layers.4.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11013120
},
{
"name": "model.layers.4.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 11016192
},
{
"name": "model.layers.4.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 11020288
},
{
"name": "model.layers.4.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 17311744
},
{
"name": "model.layers.5.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22030336
},
{
"name": "model.layers.5.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22033408
},
{
"name": "model.layers.5.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 22036480
},
{
"name": "model.layers.5.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 22040576
},
{
"name": "model.layers.5.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 28332032
},
{
"name": "model.layers.6.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33050624
},
{
"name": "model.layers.6.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33053696
},
{
"name": "model.layers.6.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 33056768
}
],
"md5sum": "e3299b1d44b3430d346c3ba55e44ebac"
},
{
"dataPath": "params_shard_59.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.7.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "3e0814820f37bbe1657146617d343337"
},
{
"dataPath": "params_shard_60.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.7.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "aacdd7799f2ddbf41c21dbebe79798b9"
},
{
"dataPath": "params_shard_61.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.8.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "6c11341814ae3390103fd06a0eb3c90e"
},
{
"dataPath": "params_shard_62.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.8.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "22f72910a0321e3237d1dd708141b6cc"
},
{
"dataPath": "params_shard_63.bin",
"format": "raw-shard",
"nbytes": 27525120,
"records": [
{
"name": "model.layers.9.mlp.down_proj.weight",
"shape": [
1536,
8960
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 27525120,
"byteOffset": 0
}
],
"md5sum": "62088d4a8bc1900127923d23a29662a1"
},
{
"dataPath": "params_shard_64.bin",
"format": "raw-shard",
"nbytes": 55050240,
"records": [
{
"name": "model.layers.9.mlp.gate_up_proj.weight",
"shape": [
17920,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 55050240,
"byteOffset": 0
}
],
"md5sum": "5d7dfcbbd08112df17ac43c388486bf0"
},
{
"dataPath": "params_shard_65.bin",
"format": "raw-shard",
"nbytes": 33060864,
"records": [
{
"name": "model.layers.6.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 0
},
{
"name": "model.layers.6.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 6291456
},
{
"name": "model.layers.7.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11010048
},
{
"name": "model.layers.7.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11013120
},
{
"name": "model.layers.7.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 11016192
},
{
"name": "model.layers.7.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 11020288
},
{
"name": "model.layers.7.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 17311744
},
{
"name": "model.layers.8.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22030336
},
{
"name": "model.layers.8.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 22033408
},
{
"name": "model.layers.8.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 22036480
},
{
"name": "model.layers.8.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 22040576
},
{
"name": "model.layers.8.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 28332032
},
{
"name": "model.layers.9.input_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33050624
},
{
"name": "model.layers.9.post_attention_layernorm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 33053696
},
{
"name": "model.layers.9.self_attn.c_attn.bias",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 33056768
}
],
"md5sum": "ad8285600a0794e1830b5778b7a30050"
},
{
"dataPath": "params_shard_66.bin",
"format": "raw-shard",
"nbytes": 11013120,
"records": [
{
"name": "model.layers.9.self_attn.c_attn.weight",
"shape": [
2048,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 6291456,
"byteOffset": 0
},
{
"name": "model.layers.9.self_attn.o_proj.weight",
"shape": [
1536,
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4718592,
"byteOffset": 6291456
},
{
"name": "model.norm.weight",
"shape": [
1536
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 3072,
"byteOffset": 11010048
}
],
"md5sum": "2cc3dc4cf44fd4ed07ad932d30783139"
}
]
}