{ "metadata": { "ParamSize": 399, "ParamBytes": 1591545856.0, "BitsPerParam": 4.1259299471863 }, "records": [ { "dataPath": "params_shard_0.bin", "format": "raw-shard", "nbytes": 155582464, "records": [ { "name": "model.embed_tokens.q_weight", "shape": [ 151936, 256 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 155582464, "byteOffset": 0 } ], "md5sum": "cf83e8ef45326a3d09a688599aa38150" }, { "dataPath": "params_shard_1.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.0.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "e2c86d4d6882f648950b44e103fbfdab" }, { "dataPath": "params_shard_2.bin", "format": "raw-shard", "nbytes": 33346560, "records": [ { "name": "model.embed_tokens.q_scale", "shape": [ 151936, 16 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4861952, "byteOffset": 0 }, { "name": "model.layers.0.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4861952 }, { "name": "model.layers.0.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4866048 }, { "name": "model.layers.0.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16138240 }, { "name": "model.layers.0.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16490496 }, { "name": "model.layers.0.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17195008 }, { "name": "model.layers.0.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17199104 }, { "name": "model.layers.0.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17204224 }, { "name": "model.layers.0.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19825664 }, { "name": "model.layers.0.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19907584 }, { "name": "model.layers.0.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22004736 }, { "name": "model.layers.1.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22070272 }, { "name": "model.layers.1.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22074368 } ], "md5sum": "4701d952ce5f93b898e6ca6cf6b77f9b" }, { "dataPath": "params_shard_3.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.1.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.1.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.1.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.1.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.1.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.1.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.1.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.1.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.1.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.10.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "fb63330a33f0e9496922a9932373fc38" }, { "dataPath": "params_shard_4.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.10.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "86ff04bc6aad76388223685a37fed73d" }, { "dataPath": "params_shard_5.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.11.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "cc8e2a34d87cd64646f44e9f457596be" }, { "dataPath": "params_shard_6.bin", "format": "raw-shard", "nbytes": 32249856, "records": [ { "name": "model.layers.10.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.10.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.10.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.10.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.10.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.10.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.10.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.10.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.10.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.11.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17204224 }, { "name": "model.layers.11.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 17208320 }, { "name": "model.layers.11.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 28480512 }, { "name": "model.layers.11.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 28832768 }, { "name": "model.layers.11.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29537280 }, { "name": "model.layers.11.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29541376 }, { "name": "model.layers.11.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 29546496 }, { "name": "model.layers.11.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 32167936 } ], "md5sum": "d98b2bf43830f4a171e60b02f2a72cdd" }, { "dataPath": "params_shard_7.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.12.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "0d4ae3f784b7eb0623e988691a1f0aef" }, { "dataPath": "params_shard_8.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.13.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "e43d2ea9e3ba32ddac03a3b1fca42acb" }, { "dataPath": "params_shard_9.bin", "format": "raw-shard", "nbytes": 31713280, "records": [ { "name": "model.layers.11.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 0 }, { "name": "model.layers.11.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 2097152 }, { "name": "model.layers.12.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 2162688 }, { "name": "model.layers.12.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 2166784 }, { "name": "model.layers.12.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 13438976 }, { "name": "model.layers.12.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 13791232 }, { "name": "model.layers.12.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 14495744 }, { "name": "model.layers.12.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14499840 }, { "name": "model.layers.12.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 14504960 }, { "name": "model.layers.12.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 17126400 }, { "name": "model.layers.12.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 17208320 }, { "name": "model.layers.12.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 19305472 }, { "name": "model.layers.13.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 19371008 }, { "name": "model.layers.13.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 19375104 }, { "name": "model.layers.13.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 30647296 }, { "name": "model.layers.13.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 30999552 }, { "name": "model.layers.13.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31704064 }, { "name": "model.layers.13.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 31708160 } ], "md5sum": "e515219cbf00da3f7ba6433339f3afd4" }, { "dataPath": "params_shard_10.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.14.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "c1fa4b8c60793e82e6dd4b36349c86b3" }, { "dataPath": "params_shard_11.bin", "format": "raw-shard", "nbytes": 33350656, "records": [ { "name": "model.layers.13.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 0 }, { "name": "model.layers.13.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 2621440 }, { "name": "model.layers.13.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 2703360 }, { "name": "model.layers.13.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 4800512 }, { "name": "model.layers.14.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4866048 }, { "name": "model.layers.14.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4870144 }, { "name": "model.layers.14.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16142336 }, { "name": "model.layers.14.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16494592 }, { "name": "model.layers.14.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17199104 }, { "name": "model.layers.14.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17203200 }, { "name": "model.layers.14.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17208320 }, { "name": "model.layers.14.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19829760 }, { "name": "model.layers.14.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19911680 }, { "name": "model.layers.14.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22008832 }, { "name": "model.layers.15.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22074368 }, { "name": "model.layers.15.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22078464 } ], "md5sum": "90e367064c183d3a81e515794a87253e" }, { "dataPath": "params_shard_12.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.15.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.15.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.15.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.15.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.15.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.15.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.15.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.15.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.15.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.16.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "c53ff2a724bf2eec98ac8a6312515e46" }, { "dataPath": "params_shard_13.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.16.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "30197075e3d9461690f083de0f6df77f" }, { "dataPath": "params_shard_14.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.17.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "39eb0747391e23e25b4c5d61425c63fe" }, { "dataPath": "params_shard_15.bin", "format": "raw-shard", "nbytes": 32249856, "records": [ { "name": "model.layers.16.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.16.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.16.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.16.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.16.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.16.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.16.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.16.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.16.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.17.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17204224 }, { "name": "model.layers.17.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 17208320 }, { "name": "model.layers.17.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 28480512 }, { "name": "model.layers.17.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 28832768 }, { "name": "model.layers.17.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29537280 }, { "name": "model.layers.17.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29541376 }, { "name": "model.layers.17.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 29546496 }, { "name": "model.layers.17.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 32167936 } ], "md5sum": "00f08771497b65279209877fd036adc3" }, { "dataPath": "params_shard_16.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.18.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "387029ff4b124ab8a7cfcaa7aa9728e9" }, { "dataPath": "params_shard_17.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.19.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "570a71c0d0baffa583bcd121bf32fa40" }, { "dataPath": "params_shard_18.bin", "format": "raw-shard", "nbytes": 31713280, "records": [ { "name": "model.layers.17.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 0 }, { "name": "model.layers.17.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 2097152 }, { "name": "model.layers.18.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 2162688 }, { "name": "model.layers.18.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 2166784 }, { "name": "model.layers.18.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 13438976 }, { "name": "model.layers.18.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 13791232 }, { "name": "model.layers.18.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 14495744 }, { "name": "model.layers.18.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14499840 }, { "name": "model.layers.18.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 14504960 }, { "name": "model.layers.18.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 17126400 }, { "name": "model.layers.18.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 17208320 }, { "name": "model.layers.18.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 19305472 }, { "name": "model.layers.19.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 19371008 }, { "name": "model.layers.19.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 19375104 }, { "name": "model.layers.19.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 30647296 }, { "name": "model.layers.19.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 30999552 }, { "name": "model.layers.19.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31704064 }, { "name": "model.layers.19.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 31708160 } ], "md5sum": "fc16887c452e02d514f8cd27527a61c3" }, { "dataPath": "params_shard_19.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.2.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "d2d1bfe650817b82ba781aaa76d67130" }, { "dataPath": "params_shard_20.bin", "format": "raw-shard", "nbytes": 33350656, "records": [ { "name": "model.layers.19.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 0 }, { "name": "model.layers.19.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 2621440 }, { "name": "model.layers.19.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 2703360 }, { "name": "model.layers.19.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 4800512 }, { "name": "model.layers.2.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4866048 }, { "name": "model.layers.2.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4870144 }, { "name": "model.layers.2.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16142336 }, { "name": "model.layers.2.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16494592 }, { "name": "model.layers.2.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17199104 }, { "name": "model.layers.2.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17203200 }, { "name": "model.layers.2.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17208320 }, { "name": "model.layers.2.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19829760 }, { "name": "model.layers.2.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19911680 }, { "name": "model.layers.2.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22008832 }, { "name": "model.layers.20.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22074368 }, { "name": "model.layers.20.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22078464 } ], "md5sum": "17beb156e9caf610423846533efa3abb" }, { "dataPath": "params_shard_21.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.20.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.20.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.20.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.20.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.20.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.20.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.20.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.20.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.20.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.21.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "91f202cf9e73abc3b6c5a4f3cdbeff42" }, { "dataPath": "params_shard_22.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.21.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "79f81e814378bc2f850f727a0b310ff7" }, { "dataPath": "params_shard_23.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.22.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "e358f184bcc18af22f63ccb7598be6e1" }, { "dataPath": "params_shard_24.bin", "format": "raw-shard", "nbytes": 32249856, "records": [ { "name": "model.layers.21.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.21.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.21.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.21.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.21.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.21.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.21.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.21.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.21.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.22.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17204224 }, { "name": "model.layers.22.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 17208320 }, { "name": "model.layers.22.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 28480512 }, { "name": "model.layers.22.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 28832768 }, { "name": "model.layers.22.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29537280 }, { "name": "model.layers.22.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29541376 }, { "name": "model.layers.22.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 29546496 }, { "name": "model.layers.22.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 32167936 } ], "md5sum": "234743eed5ef1a6bd9a49f7f2902f191" }, { "dataPath": "params_shard_25.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.23.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "6ed89d6d0e952caa655e454362a1b64d" }, { "dataPath": "params_shard_26.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.24.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "b63f571a1d1a356abf8c79ec56b761a2" }, { "dataPath": "params_shard_27.bin", "format": "raw-shard", "nbytes": 31713280, "records": [ { "name": "model.layers.22.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 0 }, { "name": "model.layers.22.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 2097152 }, { "name": "model.layers.23.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 2162688 }, { "name": "model.layers.23.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 2166784 }, { "name": "model.layers.23.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 13438976 }, { "name": "model.layers.23.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 13791232 }, { "name": "model.layers.23.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 14495744 }, { "name": "model.layers.23.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14499840 }, { "name": "model.layers.23.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 14504960 }, { "name": "model.layers.23.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 17126400 }, { "name": "model.layers.23.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 17208320 }, { "name": "model.layers.23.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 19305472 }, { "name": "model.layers.24.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 19371008 }, { "name": "model.layers.24.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 19375104 }, { "name": "model.layers.24.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 30647296 }, { "name": "model.layers.24.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 30999552 }, { "name": "model.layers.24.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31704064 }, { "name": "model.layers.24.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 31708160 } ], "md5sum": "6bf96b1bdcaebc4c1282ca5e547a78e9" }, { "dataPath": "params_shard_28.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.25.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "73f9895bbe077c2c5728b4010b21de34" }, { "dataPath": "params_shard_29.bin", "format": "raw-shard", "nbytes": 33350656, "records": [ { "name": "model.layers.24.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 0 }, { "name": "model.layers.24.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 2621440 }, { "name": "model.layers.24.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 2703360 }, { "name": "model.layers.24.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 4800512 }, { "name": "model.layers.25.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4866048 }, { "name": "model.layers.25.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4870144 }, { "name": "model.layers.25.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16142336 }, { "name": "model.layers.25.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16494592 }, { "name": "model.layers.25.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17199104 }, { "name": "model.layers.25.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17203200 }, { "name": "model.layers.25.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17208320 }, { "name": "model.layers.25.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19829760 }, { "name": "model.layers.25.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19911680 }, { "name": "model.layers.25.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22008832 }, { "name": "model.layers.26.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22074368 }, { "name": "model.layers.26.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22078464 } ], "md5sum": "0b3f268c5090a854a7d0f579afe3fb8f" }, { "dataPath": "params_shard_30.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.26.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.26.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.26.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.26.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.26.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.26.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.26.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.26.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.26.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.27.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "5e55990e6ca493d281d92be2a15629a5" }, { "dataPath": "params_shard_31.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.27.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "d93afcfbd2852982fb5ae5d8306389d3" }, { "dataPath": "params_shard_32.bin", "format": "raw-shard", "nbytes": 33351680, "records": [ { "name": "model.layers.27.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.27.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.27.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.27.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.27.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.27.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.27.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.27.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.27.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.28.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17204224 }, { "name": "model.layers.28.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17209344 }, { "name": "model.layers.28.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19830784 }, { "name": "model.layers.28.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19912704 }, { "name": "model.layers.28.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22009856 }, { "name": "model.layers.3.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22075392 }, { "name": "model.layers.3.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22079488 } ], "md5sum": "0cc56ab4118c6979ef23948e76606021" }, { "dataPath": "params_shard_33.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.3.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.3.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.3.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.3.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.3.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.3.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.3.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.3.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.3.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.4.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "1e9f29e9d6c70b654fa5ac1cb39b39b4" }, { "dataPath": "params_shard_34.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.4.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "af51cbc4ef9c3a33c0c3ca06f95a0b9d" }, { "dataPath": "params_shard_35.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.5.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "e99a7567f51fe6b95368410a1dd10fd1" }, { "dataPath": "params_shard_36.bin", "format": "raw-shard", "nbytes": 32249856, "records": [ { "name": "model.layers.4.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.4.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.4.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.4.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.4.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.4.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.4.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.4.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.4.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.5.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17204224 }, { "name": "model.layers.5.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 17208320 }, { "name": "model.layers.5.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 28480512 }, { "name": "model.layers.5.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 28832768 }, { "name": "model.layers.5.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29537280 }, { "name": "model.layers.5.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29541376 }, { "name": "model.layers.5.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 29546496 }, { "name": "model.layers.5.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 32167936 } ], "md5sum": "2482fee95789bd9a340685b1723c634b" }, { "dataPath": "params_shard_37.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.6.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "d928342db7872c81ca7ea2bc365ac177" }, { "dataPath": "params_shard_38.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.7.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "222c8da674ed99a0003148cee26a58a7" }, { "dataPath": "params_shard_39.bin", "format": "raw-shard", "nbytes": 31713280, "records": [ { "name": "model.layers.5.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 0 }, { "name": "model.layers.5.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 2097152 }, { "name": "model.layers.6.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 2162688 }, { "name": "model.layers.6.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 2166784 }, { "name": "model.layers.6.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 13438976 }, { "name": "model.layers.6.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 13791232 }, { "name": "model.layers.6.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 14495744 }, { "name": "model.layers.6.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14499840 }, { "name": "model.layers.6.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 14504960 }, { "name": "model.layers.6.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 17126400 }, { "name": "model.layers.6.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 17208320 }, { "name": "model.layers.6.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 19305472 }, { "name": "model.layers.7.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 19371008 }, { "name": "model.layers.7.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 19375104 }, { "name": "model.layers.7.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 30647296 }, { "name": "model.layers.7.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 30999552 }, { "name": "model.layers.7.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31704064 }, { "name": "model.layers.7.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 31708160 } ], "md5sum": "d8f2dff7665520129947e36b570d46fa" }, { "dataPath": "params_shard_40.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.8.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "1f59da1bb76ffd5120159dea6585b1d0" }, { "dataPath": "params_shard_41.bin", "format": "raw-shard", "nbytes": 33350656, "records": [ { "name": "model.layers.7.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 0 }, { "name": "model.layers.7.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 2621440 }, { "name": "model.layers.7.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 2703360 }, { "name": "model.layers.7.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 4800512 }, { "name": "model.layers.8.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4866048 }, { "name": "model.layers.8.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4870144 }, { "name": "model.layers.8.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16142336 }, { "name": "model.layers.8.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16494592 }, { "name": "model.layers.8.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17199104 }, { "name": "model.layers.8.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17203200 }, { "name": "model.layers.8.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17208320 }, { "name": "model.layers.8.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19829760 }, { "name": "model.layers.8.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19911680 }, { "name": "model.layers.8.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22008832 }, { "name": "model.layers.9.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22074368 }, { "name": "model.layers.9.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22078464 } ], "md5sum": "bfa8d38d1e7807ae536079b4714e9aa5" }, { "dataPath": "params_shard_42.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.9.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.9.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.9.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.9.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.9.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.9.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.9.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.9.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.9.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.28.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "a65f2495f3afc5364ac57e64a8ed6ba6" }, { "dataPath": "params_shard_43.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.28.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "697e31b491a0fda199c11083f20fe790" }, { "dataPath": "params_shard_44.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.29.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "ed08d5c51906cf3c8381cd9bb434f606" }, { "dataPath": "params_shard_45.bin", "format": "raw-shard", "nbytes": 29545472, "records": [ { "name": "model.layers.28.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.28.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.28.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.28.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.29.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12333056 }, { "name": "model.layers.29.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 12337152 }, { "name": "model.layers.29.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 23609344 }, { "name": "model.layers.29.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 23961600 }, { "name": "model.layers.29.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 24666112 }, { "name": "model.layers.29.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 24670208 }, { "name": "model.layers.29.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 24675328 }, { "name": "model.layers.29.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 27296768 }, { "name": "model.layers.29.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 27378688 }, { "name": "model.layers.29.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 29475840 }, { "name": "model.layers.30.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29541376 } ], "md5sum": "faa9a0bdb828d4572dc2e99c20db986d" }, { "dataPath": "params_shard_46.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.30.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "cfbf99928f1d8f0d07d8a1795ed9fd68" }, { "dataPath": "params_shard_47.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.31.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "e536daa218bc5654d0a24912877dacad" }, { "dataPath": "params_shard_48.bin", "format": "raw-shard", "nbytes": 32249856, "records": [ { "name": "model.layers.30.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.30.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.30.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.30.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.30.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.30.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.30.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.30.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.30.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.31.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17204224 }, { "name": "model.layers.31.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 17208320 }, { "name": "model.layers.31.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 28480512 }, { "name": "model.layers.31.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 28832768 }, { "name": "model.layers.31.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29537280 }, { "name": "model.layers.31.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29541376 }, { "name": "model.layers.31.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 29546496 }, { "name": "model.layers.31.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 32167936 } ], "md5sum": "090a0ec080ff4804b388b15ed9bebd25" }, { "dataPath": "params_shard_49.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.32.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "e90e334977c6146a833518444167f455" }, { "dataPath": "params_shard_50.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.33.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "f5df4be58c3bf9bdbe7a5436094381cb" }, { "dataPath": "params_shard_51.bin", "format": "raw-shard", "nbytes": 31713280, "records": [ { "name": "model.layers.31.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 0 }, { "name": "model.layers.31.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 2097152 }, { "name": "model.layers.32.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 2162688 }, { "name": "model.layers.32.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 2166784 }, { "name": "model.layers.32.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 13438976 }, { "name": "model.layers.32.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 13791232 }, { "name": "model.layers.32.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 14495744 }, { "name": "model.layers.32.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14499840 }, { "name": "model.layers.32.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 14504960 }, { "name": "model.layers.32.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 17126400 }, { "name": "model.layers.32.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 17208320 }, { "name": "model.layers.32.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 19305472 }, { "name": "model.layers.33.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 19371008 }, { "name": "model.layers.33.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 19375104 }, { "name": "model.layers.33.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 30647296 }, { "name": "model.layers.33.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 30999552 }, { "name": "model.layers.33.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31704064 }, { "name": "model.layers.33.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 31708160 } ], "md5sum": "b3519e961b5a72a470fe9d6b91cf5d0e" }, { "dataPath": "params_shard_52.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.34.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "850401bbda1e575253ff6ab6cedf5df5" }, { "dataPath": "params_shard_53.bin", "format": "raw-shard", "nbytes": 33350656, "records": [ { "name": "model.layers.33.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 0 }, { "name": "model.layers.33.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 2621440 }, { "name": "model.layers.33.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 2703360 }, { "name": "model.layers.33.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 4800512 }, { "name": "model.layers.34.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4866048 }, { "name": "model.layers.34.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4870144 }, { "name": "model.layers.34.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16142336 }, { "name": "model.layers.34.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16494592 }, { "name": "model.layers.34.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17199104 }, { "name": "model.layers.34.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17203200 }, { "name": "model.layers.34.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17208320 }, { "name": "model.layers.34.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19829760 }, { "name": "model.layers.34.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19911680 }, { "name": "model.layers.34.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22008832 }, { "name": "model.layers.35.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22074368 }, { "name": "model.layers.35.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22078464 } ], "md5sum": "9067b87dea9122feacbcca6b65f2c9b7" }, { "dataPath": "params_shard_54.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.35.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.35.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.35.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.35.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.35.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.35.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.35.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.35.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.35.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.norm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "63df3a31095e2d8255b2370f0aa6a647" } ] }