{ "metadata": { "ParamSize": 399, "ParamBytes": 1591545856.0, "BitsPerParam": 4.1259299471863 }, "records": [ { "dataPath": "params_shard_0.bin", "format": "raw-shard", "nbytes": 155582464, "records": [ { "name": "model.embed_tokens.q_weight", "shape": [ 151936, 256 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 155582464, "byteOffset": 0 } ], "md5sum": "3ececd062f7f5711b2909fc0e25a06a4" }, { "dataPath": "params_shard_1.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.0.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "1e18be58f2b6800eb5963989bc1f5540" }, { "dataPath": "params_shard_2.bin", "format": "raw-shard", "nbytes": 33346560, "records": [ { "name": "model.embed_tokens.q_scale", "shape": [ 151936, 16 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4861952, "byteOffset": 0 }, { "name": "model.layers.0.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4861952 }, { "name": "model.layers.0.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4866048 }, { "name": "model.layers.0.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16138240 }, { "name": "model.layers.0.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16490496 }, { "name": "model.layers.0.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17195008 }, { "name": "model.layers.0.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17199104 }, { "name": "model.layers.0.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17204224 }, { "name": "model.layers.0.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19825664 }, { "name": "model.layers.0.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19907584 }, { "name": "model.layers.0.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22004736 }, { "name": "model.layers.1.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22070272 }, { "name": "model.layers.1.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22074368 } ], "md5sum": "9946747e83d3f6b56e18246cc78c955a" }, { "dataPath": "params_shard_3.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.1.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.1.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.1.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.1.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.1.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.1.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.1.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.1.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.1.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.10.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "6097c4b2d88579f02f507c9c91b30e83" }, { "dataPath": "params_shard_4.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.10.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "37bc569c3518efd39e47e872a3981e58" }, { "dataPath": "params_shard_5.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.11.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "c527f0e3e84416adbb9bc5b745f1dcce" }, { "dataPath": "params_shard_6.bin", "format": "raw-shard", "nbytes": 32249856, "records": [ { "name": "model.layers.10.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.10.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.10.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.10.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.10.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.10.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.10.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.10.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.10.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.11.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17204224 }, { "name": "model.layers.11.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 17208320 }, { "name": "model.layers.11.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 28480512 }, { "name": "model.layers.11.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 28832768 }, { "name": "model.layers.11.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29537280 }, { "name": "model.layers.11.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29541376 }, { "name": "model.layers.11.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 29546496 }, { "name": "model.layers.11.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 32167936 } ], "md5sum": "d55070f3a19bbb4d9ea65ac836103746" }, { "dataPath": "params_shard_7.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.12.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "ab48a374626b23d42ef28df146aa80e6" }, { "dataPath": "params_shard_8.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.13.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "609c50d49beb1bb0f974635e3959ba17" }, { "dataPath": "params_shard_9.bin", "format": "raw-shard", "nbytes": 31713280, "records": [ { "name": "model.layers.11.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 0 }, { "name": "model.layers.11.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 2097152 }, { "name": "model.layers.12.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 2162688 }, { "name": "model.layers.12.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 2166784 }, { "name": "model.layers.12.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 13438976 }, { "name": "model.layers.12.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 13791232 }, { "name": "model.layers.12.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 14495744 }, { "name": "model.layers.12.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14499840 }, { "name": "model.layers.12.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 14504960 }, { "name": "model.layers.12.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 17126400 }, { "name": "model.layers.12.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 17208320 }, { "name": "model.layers.12.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 19305472 }, { "name": "model.layers.13.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 19371008 }, { "name": "model.layers.13.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 19375104 }, { "name": "model.layers.13.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 30647296 }, { "name": "model.layers.13.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 30999552 }, { "name": "model.layers.13.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31704064 }, { "name": "model.layers.13.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 31708160 } ], "md5sum": "904801f1280486b5dfe49fe89cc4f3fb" }, { "dataPath": "params_shard_10.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.14.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "cb19054457d58a1b2eb274f7f46f5050" }, { "dataPath": "params_shard_11.bin", "format": "raw-shard", "nbytes": 33350656, "records": [ { "name": "model.layers.13.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 0 }, { "name": "model.layers.13.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 2621440 }, { "name": "model.layers.13.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 2703360 }, { "name": "model.layers.13.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 4800512 }, { "name": "model.layers.14.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4866048 }, { "name": "model.layers.14.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4870144 }, { "name": "model.layers.14.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16142336 }, { "name": "model.layers.14.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16494592 }, { "name": "model.layers.14.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17199104 }, { "name": "model.layers.14.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17203200 }, { "name": "model.layers.14.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17208320 }, { "name": "model.layers.14.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19829760 }, { "name": "model.layers.14.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19911680 }, { "name": "model.layers.14.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22008832 }, { "name": "model.layers.15.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22074368 }, { "name": "model.layers.15.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22078464 } ], "md5sum": "d3c4ff4c480bba99a73df4cdbbb8da0a" }, { "dataPath": "params_shard_12.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.15.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.15.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.15.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.15.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.15.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.15.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.15.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.15.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.15.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.16.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "4360d997dbcd82b53546f33a92ce6c9d" }, { "dataPath": "params_shard_13.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.16.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "ab65b534fb8a0d0665409cf6f919a8f9" }, { "dataPath": "params_shard_14.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.17.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "c9250717a9603c836e71254313c158ac" }, { "dataPath": "params_shard_15.bin", "format": "raw-shard", "nbytes": 32249856, "records": [ { "name": "model.layers.16.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.16.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.16.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.16.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.16.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.16.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.16.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.16.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.16.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.17.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17204224 }, { "name": "model.layers.17.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 17208320 }, { "name": "model.layers.17.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 28480512 }, { "name": "model.layers.17.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 28832768 }, { "name": "model.layers.17.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29537280 }, { "name": "model.layers.17.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29541376 }, { "name": "model.layers.17.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 29546496 }, { "name": "model.layers.17.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 32167936 } ], "md5sum": "411a6c805a3b55dac858304227ec6977" }, { "dataPath": "params_shard_16.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.18.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "f684cc11733840f737329e127a403a1a" }, { "dataPath": "params_shard_17.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.19.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "65dcf4fb055c2a11174aca029efe40a0" }, { "dataPath": "params_shard_18.bin", "format": "raw-shard", "nbytes": 31713280, "records": [ { "name": "model.layers.17.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 0 }, { "name": "model.layers.17.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 2097152 }, { "name": "model.layers.18.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 2162688 }, { "name": "model.layers.18.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 2166784 }, { "name": "model.layers.18.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 13438976 }, { "name": "model.layers.18.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 13791232 }, { "name": "model.layers.18.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 14495744 }, { "name": "model.layers.18.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14499840 }, { "name": "model.layers.18.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 14504960 }, { "name": "model.layers.18.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 17126400 }, { "name": "model.layers.18.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 17208320 }, { "name": "model.layers.18.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 19305472 }, { "name": "model.layers.19.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 19371008 }, { "name": "model.layers.19.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 19375104 }, { "name": "model.layers.19.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 30647296 }, { "name": "model.layers.19.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 30999552 }, { "name": "model.layers.19.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31704064 }, { "name": "model.layers.19.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 31708160 } ], "md5sum": "049a01bb65e49b17113f477ede112fda" }, { "dataPath": "params_shard_19.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.2.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "471bfcb70d002350f30c26eb6649ba8d" }, { "dataPath": "params_shard_20.bin", "format": "raw-shard", "nbytes": 33350656, "records": [ { "name": "model.layers.19.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 0 }, { "name": "model.layers.19.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 2621440 }, { "name": "model.layers.19.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 2703360 }, { "name": "model.layers.19.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 4800512 }, { "name": "model.layers.2.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4866048 }, { "name": "model.layers.2.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4870144 }, { "name": "model.layers.2.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16142336 }, { "name": "model.layers.2.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16494592 }, { "name": "model.layers.2.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17199104 }, { "name": "model.layers.2.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17203200 }, { "name": "model.layers.2.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17208320 }, { "name": "model.layers.2.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19829760 }, { "name": "model.layers.2.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19911680 }, { "name": "model.layers.2.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22008832 }, { "name": "model.layers.20.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22074368 }, { "name": "model.layers.20.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22078464 } ], "md5sum": "5b4047cdacc35a00a44f3962f0a455d1" }, { "dataPath": "params_shard_21.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.20.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.20.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.20.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.20.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.20.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.20.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.20.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.20.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.20.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.21.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "7608b0f67fa92f1f8e0f868d09f9e7fa" }, { "dataPath": "params_shard_22.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.21.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "b2c88f8dbd4a7078f38a97eea975579e" }, { "dataPath": "params_shard_23.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.22.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "5f5e6a60d1710725f1259fe77bca9f49" }, { "dataPath": "params_shard_24.bin", "format": "raw-shard", "nbytes": 32249856, "records": [ { "name": "model.layers.21.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.21.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.21.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.21.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.21.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.21.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.21.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.21.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.21.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.22.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17204224 }, { "name": "model.layers.22.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 17208320 }, { "name": "model.layers.22.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 28480512 }, { "name": "model.layers.22.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 28832768 }, { "name": "model.layers.22.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29537280 }, { "name": "model.layers.22.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29541376 }, { "name": "model.layers.22.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 29546496 }, { "name": "model.layers.22.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 32167936 } ], "md5sum": "54af68cd87da19d1da2bcf3cd1e90c61" }, { "dataPath": "params_shard_25.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.23.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "0322c121bcbcbcad09b931a4a5141f40" }, { "dataPath": "params_shard_26.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.24.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "749e73d18d16f561157d3f578fd32b1e" }, { "dataPath": "params_shard_27.bin", "format": "raw-shard", "nbytes": 31713280, "records": [ { "name": "model.layers.22.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 0 }, { "name": "model.layers.22.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 2097152 }, { "name": "model.layers.23.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 2162688 }, { "name": "model.layers.23.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 2166784 }, { "name": "model.layers.23.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 13438976 }, { "name": "model.layers.23.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 13791232 }, { "name": "model.layers.23.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 14495744 }, { "name": "model.layers.23.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14499840 }, { "name": "model.layers.23.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 14504960 }, { "name": "model.layers.23.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 17126400 }, { "name": "model.layers.23.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 17208320 }, { "name": "model.layers.23.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 19305472 }, { "name": "model.layers.24.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 19371008 }, { "name": "model.layers.24.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 19375104 }, { "name": "model.layers.24.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 30647296 }, { "name": "model.layers.24.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 30999552 }, { "name": "model.layers.24.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31704064 }, { "name": "model.layers.24.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 31708160 } ], "md5sum": "416922d84c25162e38e5f7f399aa8665" }, { "dataPath": "params_shard_28.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.25.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "904c77f32952a8b647365df68f9b29f3" }, { "dataPath": "params_shard_29.bin", "format": "raw-shard", "nbytes": 33350656, "records": [ { "name": "model.layers.24.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 0 }, { "name": "model.layers.24.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 2621440 }, { "name": "model.layers.24.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 2703360 }, { "name": "model.layers.24.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 4800512 }, { "name": "model.layers.25.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4866048 }, { "name": "model.layers.25.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4870144 }, { "name": "model.layers.25.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16142336 }, { "name": "model.layers.25.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16494592 }, { "name": "model.layers.25.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17199104 }, { "name": "model.layers.25.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17203200 }, { "name": "model.layers.25.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17208320 }, { "name": "model.layers.25.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19829760 }, { "name": "model.layers.25.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19911680 }, { "name": "model.layers.25.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22008832 }, { "name": "model.layers.26.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22074368 }, { "name": "model.layers.26.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22078464 } ], "md5sum": "3d5e2617d6e443cce861f19bdb0ccc80" }, { "dataPath": "params_shard_30.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.26.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.26.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.26.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.26.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.26.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.26.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.26.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.26.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.26.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.27.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "f4b2383389b1816985343607f4791c69" }, { "dataPath": "params_shard_31.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.27.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "46a0af2d62c143e48161cbc239d1a4c8" }, { "dataPath": "params_shard_32.bin", "format": "raw-shard", "nbytes": 33351680, "records": [ { "name": "model.layers.27.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.27.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.27.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.27.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.27.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.27.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.27.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.27.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.27.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.28.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17204224 }, { "name": "model.layers.28.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17209344 }, { "name": "model.layers.28.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19830784 }, { "name": "model.layers.28.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19912704 }, { "name": "model.layers.28.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22009856 }, { "name": "model.layers.3.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22075392 }, { "name": "model.layers.3.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22079488 } ], "md5sum": "11b829f1063b535514f6989f47f98018" }, { "dataPath": "params_shard_33.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.3.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.3.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.3.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.3.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.3.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.3.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.3.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.3.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.3.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.4.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "2cb957956396c3fd1003fa83e694d89a" }, { "dataPath": "params_shard_34.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.4.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "71e7f3ddde6abec6b1fc8ea4e3c1478a" }, { "dataPath": "params_shard_35.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.5.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "fc40673a3f81fcca2563b6b0fc6a9e2d" }, { "dataPath": "params_shard_36.bin", "format": "raw-shard", "nbytes": 32249856, "records": [ { "name": "model.layers.4.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.4.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.4.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.4.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.4.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.4.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.4.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.4.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.4.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.5.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17204224 }, { "name": "model.layers.5.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 17208320 }, { "name": "model.layers.5.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 28480512 }, { "name": "model.layers.5.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 28832768 }, { "name": "model.layers.5.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29537280 }, { "name": "model.layers.5.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29541376 }, { "name": "model.layers.5.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 29546496 }, { "name": "model.layers.5.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 32167936 } ], "md5sum": "4913d8e7e6dfd4e03bc62cc3a3f3af70" }, { "dataPath": "params_shard_37.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.6.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "101a33d1df99ef613b501eafbc0fd736" }, { "dataPath": "params_shard_38.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.7.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "c35bdd4841745e048b0e1a3ab84e64d6" }, { "dataPath": "params_shard_39.bin", "format": "raw-shard", "nbytes": 31713280, "records": [ { "name": "model.layers.5.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 0 }, { "name": "model.layers.5.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 2097152 }, { "name": "model.layers.6.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 2162688 }, { "name": "model.layers.6.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 2166784 }, { "name": "model.layers.6.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 13438976 }, { "name": "model.layers.6.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 13791232 }, { "name": "model.layers.6.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 14495744 }, { "name": "model.layers.6.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14499840 }, { "name": "model.layers.6.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 14504960 }, { "name": "model.layers.6.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 17126400 }, { "name": "model.layers.6.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 17208320 }, { "name": "model.layers.6.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 19305472 }, { "name": "model.layers.7.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 19371008 }, { "name": "model.layers.7.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 19375104 }, { "name": "model.layers.7.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 30647296 }, { "name": "model.layers.7.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 30999552 }, { "name": "model.layers.7.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31704064 }, { "name": "model.layers.7.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 31708160 } ], "md5sum": "2d8fc7887941544c8a40dfa87619a207" }, { "dataPath": "params_shard_40.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.8.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "1d93f46ba8e75d12d01117bc0f7229bc" }, { "dataPath": "params_shard_41.bin", "format": "raw-shard", "nbytes": 33350656, "records": [ { "name": "model.layers.7.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 0 }, { "name": "model.layers.7.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 2621440 }, { "name": "model.layers.7.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 2703360 }, { "name": "model.layers.7.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 4800512 }, { "name": "model.layers.8.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4866048 }, { "name": "model.layers.8.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4870144 }, { "name": "model.layers.8.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16142336 }, { "name": "model.layers.8.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16494592 }, { "name": "model.layers.8.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17199104 }, { "name": "model.layers.8.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17203200 }, { "name": "model.layers.8.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17208320 }, { "name": "model.layers.8.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19829760 }, { "name": "model.layers.8.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19911680 }, { "name": "model.layers.8.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22008832 }, { "name": "model.layers.9.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22074368 }, { "name": "model.layers.9.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22078464 } ], "md5sum": "848abdd4598d4af16de5aaaff21de555" }, { "dataPath": "params_shard_42.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.9.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.9.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.9.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.9.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.9.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.9.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.9.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.9.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.9.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.layers.28.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "01abec07d2d11d6409644d1cb163d338" }, { "dataPath": "params_shard_43.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.28.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "1f38e7e76c5680bae70b7bdf4361735d" }, { "dataPath": "params_shard_44.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.29.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "3a75416db27e1a9f9bb2b398e34b96b7" }, { "dataPath": "params_shard_45.bin", "format": "raw-shard", "nbytes": 29545472, "records": [ { "name": "model.layers.28.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.28.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.28.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.28.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.29.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12333056 }, { "name": "model.layers.29.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 12337152 }, { "name": "model.layers.29.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 23609344 }, { "name": "model.layers.29.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 23961600 }, { "name": "model.layers.29.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 24666112 }, { "name": "model.layers.29.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 24670208 }, { "name": "model.layers.29.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 24675328 }, { "name": "model.layers.29.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 27296768 }, { "name": "model.layers.29.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 27378688 }, { "name": "model.layers.29.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 29475840 }, { "name": "model.layers.30.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29541376 } ], "md5sum": "9ef06429510cd85a794b9ee85bc274a4" }, { "dataPath": "params_shard_46.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.30.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "d169d089da7079f1643554e39be3bd1c" }, { "dataPath": "params_shard_47.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.31.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "13db058066d533137ccab06ba0d34d7c" }, { "dataPath": "params_shard_48.bin", "format": "raw-shard", "nbytes": 32249856, "records": [ { "name": "model.layers.30.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 0 }, { "name": "model.layers.30.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 11272192 }, { "name": "model.layers.30.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 11624448 }, { "name": "model.layers.30.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 12328960 }, { "name": "model.layers.30.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 12333056 }, { "name": "model.layers.30.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 12338176 }, { "name": "model.layers.30.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 14959616 }, { "name": "model.layers.30.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 15041536 }, { "name": "model.layers.30.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 17138688 }, { "name": "model.layers.31.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17204224 }, { "name": "model.layers.31.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 17208320 }, { "name": "model.layers.31.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 28480512 }, { "name": "model.layers.31.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 28832768 }, { "name": "model.layers.31.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 29537280 }, { "name": "model.layers.31.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29541376 }, { "name": "model.layers.31.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 29546496 }, { "name": "model.layers.31.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 32167936 } ], "md5sum": "902888821984031670f8df922a8c427a" }, { "dataPath": "params_shard_49.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.32.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "66c212ed6a920a59456153742c9384ec" }, { "dataPath": "params_shard_50.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.33.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "fa8094b05344885d20016e890174726d" }, { "dataPath": "params_shard_51.bin", "format": "raw-shard", "nbytes": 31713280, "records": [ { "name": "model.layers.31.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 0 }, { "name": "model.layers.31.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 2097152 }, { "name": "model.layers.32.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 2162688 }, { "name": "model.layers.32.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 2166784 }, { "name": "model.layers.32.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 13438976 }, { "name": "model.layers.32.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 13791232 }, { "name": "model.layers.32.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 14495744 }, { "name": "model.layers.32.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14499840 }, { "name": "model.layers.32.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 14504960 }, { "name": "model.layers.32.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 17126400 }, { "name": "model.layers.32.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 17208320 }, { "name": "model.layers.32.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 19305472 }, { "name": "model.layers.33.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 19371008 }, { "name": "model.layers.33.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 19375104 }, { "name": "model.layers.33.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 30647296 }, { "name": "model.layers.33.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 30999552 }, { "name": "model.layers.33.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31704064 }, { "name": "model.layers.33.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 31708160 } ], "md5sum": "192480cd45a55976eca3e014a3083cb6" }, { "dataPath": "params_shard_52.bin", "format": "raw-shard", "nbytes": 22544384, "records": [ { "name": "model.layers.34.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 0 } ], "md5sum": "78b671bd60c5d522ccc00ab21e5af389" }, { "dataPath": "params_shard_53.bin", "format": "raw-shard", "nbytes": 33350656, "records": [ { "name": "model.layers.33.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 0 }, { "name": "model.layers.33.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 2621440 }, { "name": "model.layers.33.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 2703360 }, { "name": "model.layers.33.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 4800512 }, { "name": "model.layers.34.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4866048 }, { "name": "model.layers.34.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 4870144 }, { "name": "model.layers.34.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 16142336 }, { "name": "model.layers.34.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 16494592 }, { "name": "model.layers.34.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 17199104 }, { "name": "model.layers.34.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 17203200 }, { "name": "model.layers.34.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 17208320 }, { "name": "model.layers.34.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 19829760 }, { "name": "model.layers.34.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 19911680 }, { "name": "model.layers.34.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 22008832 }, { "name": "model.layers.35.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22074368 }, { "name": "model.layers.35.mlp.down_proj.q_weight", "shape": [ 1376, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 11272192, "byteOffset": 22078464 } ], "md5sum": "4d8b72d7de184211d063e597222f786f" }, { "dataPath": "params_shard_54.bin", "format": "raw-shard", "nbytes": 28480512, "records": [ { "name": "model.layers.35.mlp.down_proj.q_scale", "shape": [ 86, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 352256, "byteOffset": 0 }, { "name": "model.layers.35.mlp.gate_up_proj.q_weight", "shape": [ 256, 22016 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 22544384, "byteOffset": 352256 }, { "name": "model.layers.35.mlp.gate_up_proj.q_scale", "shape": [ 16, 22016 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 704512, "byteOffset": 22896640 }, { "name": "model.layers.35.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23601152 }, { "name": "model.layers.35.self_attn.c_attn.bias", "shape": [ 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 23605248 }, { "name": "model.layers.35.self_attn.c_attn.q_weight", "shape": [ 256, 2560 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2621440, "byteOffset": 23610368 }, { "name": "model.layers.35.self_attn.c_attn.q_scale", "shape": [ 16, 2560 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 81920, "byteOffset": 26231808 }, { "name": "model.layers.35.self_attn.o_proj.q_weight", "shape": [ 256, 2048 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 2097152, "byteOffset": 26313728 }, { "name": "model.layers.35.self_attn.o_proj.q_scale", "shape": [ 16, 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 65536, "byteOffset": 28410880 }, { "name": "model.norm.weight", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 28476416 } ], "md5sum": "cbdbf3b59c2432158b8e60195686e755" } ] }