{ "metadata": { "ParamSize": 709, "ParamBytes": 16893428384.0, "BitsPerParam": 4.125405707111059 }, "records": [ { "dataPath": "params_shard_0.bin", "format": "raw-shard", "nbytes": 388262400, "records": [ { "name": "lm_head.q_weight", "shape": [ 640, 151665 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 388262400, "byteOffset": 0 } ], "md5sum": "de24ba93c85aa0dc384b3dc31ac2e8c5" }, { "dataPath": "params_shard_1.bin", "format": "raw-shard", "nbytes": 388262400, "records": [ { "name": "model.embed_tokens.q_weight", "shape": [ 151665, 640 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 388262400, "byteOffset": 0 } ], "md5sum": "042009f6a119512d47484ca8828d8181" }, { "dataPath": "params_shard_2.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.0.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "011be5df2eb27d62f10e162aae176289" }, { "dataPath": "params_shard_3.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.0.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "a768dfbe90169ffeeeb7852ac0db0f00" }, { "dataPath": "params_shard_4.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.0.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "dde7436001ccfcc7b355c8b5b36e2195" }, { "dataPath": "params_shard_5.bin", "format": "raw-shard", "nbytes": 31510176, "records": [ { "name": "lm_head.q_scale", "shape": [ 40, 151665 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 12133200, "byteOffset": 0 }, { "name": "model.embed_tokens.q_scale", "shape": [ 151665, 40 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 12133200, "byteOffset": 12133200 }, { "name": "model.layers.0.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 24266400 }, { "name": "model.layers.0.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 24276640 }, { "name": "model.layers.0.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 26488480 }, { "name": "model.layers.0.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 30912160 }, { "name": "model.layers.0.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 30922400 }, { "name": "model.layers.0.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 30936736 } ], "md5sum": "6b143f491a82dde1def1325117beb4fd" }, { "dataPath": "params_shard_6.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.1.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "90ea2bdc74b8cb2003245abbc5c787d6" }, { "dataPath": "params_shard_7.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.1.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "a13366baba201901ad52455181378ced" }, { "dataPath": "params_shard_8.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.1.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "d695ce78c9ab2793e156709721058f16" }, { "dataPath": "params_shard_9.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.0.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.0.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.1.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.1.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.1.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.1.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.1.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.1.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "88dd3962f8eaa5b6b0a3eef3dfbab202" }, { "dataPath": "params_shard_10.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.10.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "9b5eb91cf59aab63e07971f6c68fd223" }, { "dataPath": "params_shard_11.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.10.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "2121e59e50226e12763456e5fdfdcc0e" }, { "dataPath": "params_shard_12.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.10.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "5ff34211392c4e4fa16176ad9a13cf14" }, { "dataPath": "params_shard_13.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.1.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.1.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.10.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.10.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.10.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.10.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.10.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.10.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "e8784018a8fef879c6283d56a9a4a9c1" }, { "dataPath": "params_shard_14.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.11.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "e65e0f75088ebf2646b4fa3563813ea0" }, { "dataPath": "params_shard_15.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.11.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "a134bd97c47dd3deb58d4e9484c0ca84" }, { "dataPath": "params_shard_16.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.11.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "07789465c7c39b294218ef49c7d38e41" }, { "dataPath": "params_shard_17.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.10.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.10.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.11.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.11.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.11.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.11.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.11.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.11.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "44c54b92417cef34dc3e8483aea7fb5c" }, { "dataPath": "params_shard_18.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.12.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "0a83dd6bd36f5b56f834c8d117900180" }, { "dataPath": "params_shard_19.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.12.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "4daa52779e94249ab13eb1fcb4d89715" }, { "dataPath": "params_shard_20.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.12.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "3677e588ebfc6a6eaffc8940404bd1d5" }, { "dataPath": "params_shard_21.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.11.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.11.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.12.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.12.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.12.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.12.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.12.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.12.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "013e986e5cc47fae99ad142d19fde38f" }, { "dataPath": "params_shard_22.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.13.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "f057d08e0831537c3441248dc4f2fe78" }, { "dataPath": "params_shard_23.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.13.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "598555c6511f3185d7f10d954ded3159" }, { "dataPath": "params_shard_24.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.13.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "9490828339d61bb440b2a6993636e5fc" }, { "dataPath": "params_shard_25.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.12.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.12.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.13.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.13.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.13.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.13.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.13.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.13.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "5f507b884b1b2c3c500822caa710a260" }, { "dataPath": "params_shard_26.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.14.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "ae747d263e4d994938aa85d8d5da0b12" }, { "dataPath": "params_shard_27.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.14.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "6cd7827c8056777d2a4fd3f37d622ced" }, { "dataPath": "params_shard_28.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.14.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "3fda2d535c73d9e105662a4a617ad3fd" }, { "dataPath": "params_shard_29.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.13.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.13.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.14.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.14.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.14.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.14.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.14.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.14.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "fbe559dc332e1a5f7980cce37b1b969b" }, { "dataPath": "params_shard_30.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.15.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "218017e614ba2895e611aa0dd2b4c4cc" }, { "dataPath": "params_shard_31.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.15.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "bc9d402c4bac88878becd63a5382db6a" }, { "dataPath": "params_shard_32.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.15.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "1596479621c3ea6438d3e41515a53a6a" }, { "dataPath": "params_shard_33.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.14.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.14.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.15.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.15.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.15.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.15.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.15.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.15.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "e2cc07a85c3014d0ebc9d5d59fe05d1f" }, { "dataPath": "params_shard_34.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.16.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "8727a38f2e975cfc64e4df2c75dad6c5" }, { "dataPath": "params_shard_35.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.16.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "5409f48d05777feb3b81ac26f5887f37" }, { "dataPath": "params_shard_36.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.16.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "9091f0b8c8cbc07707bc208cc656d32d" }, { "dataPath": "params_shard_37.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.15.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.15.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.16.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.16.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.16.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.16.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.16.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.16.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "253af6c09ade772f2a95bf0bae27dd07" }, { "dataPath": "params_shard_38.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.17.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "fcf5c3d57befbfa114e658c8079e3a6e" }, { "dataPath": "params_shard_39.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.17.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "1874557430ec0ae7eb516dcf1dc4e1a1" }, { "dataPath": "params_shard_40.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.17.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "32a351903d777eef85f35f30d87fb11f" }, { "dataPath": "params_shard_41.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.16.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.16.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.17.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.17.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.17.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.17.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.17.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.17.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "d84e6e081737434fa1147a8970bd2771" }, { "dataPath": "params_shard_42.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.18.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "27f2094050c58eb9400805a8c13be1fb" }, { "dataPath": "params_shard_43.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.18.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "4f5ec70c1d10b329b0dda76f55e3e7f7" }, { "dataPath": "params_shard_44.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.18.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "d43d520ff6bf2d96dadb6f39988f6099" }, { "dataPath": "params_shard_45.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.17.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.17.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.18.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.18.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.18.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.18.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.18.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.18.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "9183b905e6a10b9fa17b1bbb0c70c930" }, { "dataPath": "params_shard_46.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.19.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "bba67a7326b6d2864cfa82fcb66f9f3f" }, { "dataPath": "params_shard_47.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.19.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "ea7f025b75db7532f3f6946ddff616d8" }, { "dataPath": "params_shard_48.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.19.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "185386f49a513ad9a45b9e6ad74f9011" }, { "dataPath": "params_shard_49.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.18.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.18.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.19.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.19.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.19.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.19.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.19.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.19.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "f668a67a832ddf8ef6e18fa011e34ac5" }, { "dataPath": "params_shard_50.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.2.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "c57c6696f3a4b5848f33e83b8a350226" }, { "dataPath": "params_shard_51.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.2.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "f2507ecbd269b2d75a713c794bdad97e" }, { "dataPath": "params_shard_52.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.2.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "66c6406e3cf3a37c3c4d710104a63f33" }, { "dataPath": "params_shard_53.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.19.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.19.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.2.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.2.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.2.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.2.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.2.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.2.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "7c40560232f312b243d7d912d369f6ad" }, { "dataPath": "params_shard_54.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.20.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "6687f6e8b747a951af4cfec37f9192bc" }, { "dataPath": "params_shard_55.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.20.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "814a4aad76d3e24436be0013359d823b" }, { "dataPath": "params_shard_56.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.20.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "6ad19d574e015af533adfb4dcd0ad7a3" }, { "dataPath": "params_shard_57.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.2.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.2.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.20.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.20.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.20.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.20.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.20.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.20.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "8ae6f493eb685e5a099e29ae122b4bf2" }, { "dataPath": "params_shard_58.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.21.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "cab14c88fdf5380dc651634e21390fa8" }, { "dataPath": "params_shard_59.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.21.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "bc87fbe610dffe9d4796a6b8735c2891" }, { "dataPath": "params_shard_60.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.21.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "56bdf6fa5b5a5b6b898298c012e87625" }, { "dataPath": "params_shard_61.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.20.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.20.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.21.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.21.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.21.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.21.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.21.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.21.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "05d9162212559e16a042b25a6903856f" }, { "dataPath": "params_shard_62.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.22.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "fd7bc0a911d22731991510e5015e8487" }, { "dataPath": "params_shard_63.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.22.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "72c81faa4f1c182b03a98197c610f325" }, { "dataPath": "params_shard_64.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.22.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "7ddca10fa414054d68343601a50c6e40" }, { "dataPath": "params_shard_65.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.21.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.21.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.22.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.22.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.22.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.22.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.22.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.22.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "f849d241c0cec65a4c79caee2c3eaa49" }, { "dataPath": "params_shard_66.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.23.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "ef02d8a5a024b3f26c6144a5ab1ac197" }, { "dataPath": "params_shard_67.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.23.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "6b3146aa00a2e24e50c6ae8f2d3d9ff7" }, { "dataPath": "params_shard_68.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.23.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "5836e05189e3e5ae2cc421dae0184b2e" }, { "dataPath": "params_shard_69.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.22.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.22.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.23.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.23.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.23.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.23.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.23.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.23.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "06c910758bab79405922a5b9e1c21e69" }, { "dataPath": "params_shard_70.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.24.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "522739b039c0de853fcdfe11c6ace080" }, { "dataPath": "params_shard_71.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.24.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "9389c4a3a345273490ad1b3b841b9b1e" }, { "dataPath": "params_shard_72.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.24.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "dabde449792c5de23f2a5d1c047093b6" }, { "dataPath": "params_shard_73.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.23.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.23.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.24.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.24.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.24.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.24.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.24.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.24.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "43ae9121a919d4b577c398acb3b592c5" }, { "dataPath": "params_shard_74.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.25.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "fab25f03b9963069e2423bcefe49ed00" }, { "dataPath": "params_shard_75.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.25.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "26e45c3ce5af27c37be181dbfddac9de" }, { "dataPath": "params_shard_76.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.25.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "47549a97508963917a1276edf345c498" }, { "dataPath": "params_shard_77.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.24.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.24.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.25.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.25.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.25.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.25.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.25.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.25.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "557574dcfff08404b0c726c9af2d1ede" }, { "dataPath": "params_shard_78.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.26.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "a2e747db83db1b92bd7e99a460da51c2" }, { "dataPath": "params_shard_79.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.26.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "f68912fc07fce51dec9e3613a64a75ab" }, { "dataPath": "params_shard_80.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.26.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "50152acad89f0ee32efd2de00074ff63" }, { "dataPath": "params_shard_81.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.25.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.25.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.26.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.26.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.26.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.26.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.26.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.26.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "04f2bc758f4bf333ce6fc93a39f89c26" }, { "dataPath": "params_shard_82.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.27.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "c0888734711d88a18b88ebed02fc99c8" }, { "dataPath": "params_shard_83.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.27.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "cd1c8e4840fcffbbb3e02891a18e5cc2" }, { "dataPath": "params_shard_84.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.27.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "ea4f216842c0ac55fca401ab8d495355" }, { "dataPath": "params_shard_85.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.26.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.26.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.27.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.27.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.27.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.27.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.27.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.27.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "6bd9c2c4b9f0a29245e55b80fea123fb" }, { "dataPath": "params_shard_86.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.28.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "8220c06603d976a9089bc666c0dd7ecf" }, { "dataPath": "params_shard_87.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.28.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "15c37f25084b5ce2d639309e9f2dbab2" }, { "dataPath": "params_shard_88.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.28.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "90ef02a472d06d92ae76e370663cf93b" }, { "dataPath": "params_shard_89.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.27.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.27.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.28.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.28.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.28.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.28.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.28.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.28.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "7323e3b86c3c06cb20eac450f6d568e2" }, { "dataPath": "params_shard_90.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.29.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "f26f40283d7c91625ede8fa234618c93" }, { "dataPath": "params_shard_91.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.29.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "48adf5ee7a35d6139cc5fe04845cee73" }, { "dataPath": "params_shard_92.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.29.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "17f6f6c28ec7c0b7442430e30a75df9d" }, { "dataPath": "params_shard_93.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.28.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.28.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.29.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.29.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.29.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.29.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.29.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.29.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "b14ac5417f9d2d98b7f9df9f8229dcbe" }, { "dataPath": "params_shard_94.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.3.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "d9ed100d930d10b9417f72138c6683f0" }, { "dataPath": "params_shard_95.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.3.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "bf59047be12de7e6e4997d6279d4d21a" }, { "dataPath": "params_shard_96.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.3.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "964db03de04f51ae5e8a4d458e04ec78" }, { "dataPath": "params_shard_97.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.29.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.29.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.3.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.3.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.3.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.3.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.3.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.3.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "ee875012ffd013797d1d320e85a5c4ad" }, { "dataPath": "params_shard_98.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.30.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "394cb7a1f42726fde6a583d58371e54e" }, { "dataPath": "params_shard_99.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.30.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "1af192537e5de095df22cd22ef4771b3" }, { "dataPath": "params_shard_100.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.30.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "0ac683584daf7626dac460c719713594" }, { "dataPath": "params_shard_101.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.3.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.3.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.30.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.30.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.30.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.30.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.30.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.30.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "02642c6d8614712eef906c9e06bad440" }, { "dataPath": "params_shard_102.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.31.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "50a5175c7af139b1fd4dcb78ff72942b" }, { "dataPath": "params_shard_103.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.31.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "195d9abdfe6b0b501d4a500815ffe0eb" }, { "dataPath": "params_shard_104.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.31.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "461bec58442de0ff569039d6af014e58" }, { "dataPath": "params_shard_105.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.30.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.30.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.31.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.31.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.31.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.31.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.31.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.31.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "473b181bb0d28f1417d9ea00142eac23" }, { "dataPath": "params_shard_106.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.32.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "1940c349cfc3401e2bbdc6762b90a1be" }, { "dataPath": "params_shard_107.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.32.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "5e11d241962cf49ce1c4e7806e98ef36" }, { "dataPath": "params_shard_108.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.32.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "f520a622e7446c83399b54dfdf3045c4" }, { "dataPath": "params_shard_109.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.31.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.31.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.32.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.32.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.32.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.32.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.32.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.32.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "5a25407f18427d6a5031e728e09e5556" }, { "dataPath": "params_shard_110.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.33.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "2931b93edea51d03e2e68cc50e2a96b7" }, { "dataPath": "params_shard_111.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.33.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "ccc58fe91ac29980850fbc2ba3a95482" }, { "dataPath": "params_shard_112.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.33.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "73f6f6d31465178a606aa95c9ac764ff" }, { "dataPath": "params_shard_113.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.32.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.32.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.33.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.33.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.33.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.33.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.33.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.33.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "a41c13b2973ed0591b21055c1923a6f1" }, { "dataPath": "params_shard_114.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.34.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "e6a20e0f30e1d711c0702f90099b32e5" }, { "dataPath": "params_shard_115.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.34.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "7ae12cb17ba73d913dbb4f559e45a65a" }, { "dataPath": "params_shard_116.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.34.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "03192bab2b47710dd5d63468279c22d1" }, { "dataPath": "params_shard_117.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.33.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.33.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.34.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.34.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.34.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.34.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.34.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.34.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "85acbcbf51ef3b9c7cded26ee627fa55" }, { "dataPath": "params_shard_118.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.35.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "d3092eab9ecff1fe5e019c93b2e46618" }, { "dataPath": "params_shard_119.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.35.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "2883f506862066f8fba6d7a01aa07ddd" }, { "dataPath": "params_shard_120.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.35.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "0ed9c3c93690ff3350d4840387289e80" }, { "dataPath": "params_shard_121.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.34.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.34.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.35.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.35.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.35.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.35.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.35.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.35.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "cf6829018a0c5fb10582919b42242ad2" }, { "dataPath": "params_shard_122.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.36.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "f84d38d15629d39a8a64f2581a8cfe76" }, { "dataPath": "params_shard_123.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.36.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "c0c8e7c86dd66a08739113b8a5668396" }, { "dataPath": "params_shard_124.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.36.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "7dee5054b899575d4dd1435b9a387f2b" }, { "dataPath": "params_shard_125.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.35.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.35.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.36.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.36.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.36.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.36.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.36.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.36.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "c1041d6afe7ba71b1e6887d921f80e99" }, { "dataPath": "params_shard_126.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.37.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "b9e4d3dbc60c689294903314d1e2ecaa" }, { "dataPath": "params_shard_127.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.37.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "a9a2e386e4fa5d940d66caef2580ad2c" }, { "dataPath": "params_shard_128.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.37.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "d9f95d2166235e2646d531e533d7f719" }, { "dataPath": "params_shard_129.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.36.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.36.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.37.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.37.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.37.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.37.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.37.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.37.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "fe68ab6269ef661057419d92549963b4" }, { "dataPath": "params_shard_130.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.38.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "a4e42dac2023bf21f816982dbfbaa941" }, { "dataPath": "params_shard_131.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.38.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "e921e46334887c93203964cf434372d1" }, { "dataPath": "params_shard_132.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.38.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "766dc3c406829eaed1d25b3880ebb0ae" }, { "dataPath": "params_shard_133.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.37.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.37.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.38.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.38.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.38.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.38.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.38.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.38.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "d6bbd9e07152506a2445e0fa9a1025f5" }, { "dataPath": "params_shard_134.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.39.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "b671bb13131ba4f8ba021089419afa53" }, { "dataPath": "params_shard_135.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.39.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "15e86b336f9c764db95153cac57c72b5" }, { "dataPath": "params_shard_136.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.39.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "b68ea780cfeec5560f7db910c38ad4dc" }, { "dataPath": "params_shard_137.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.38.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.38.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.39.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.39.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.39.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.39.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.39.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.39.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "4048c5e393924631dd932d64ed80a813" }, { "dataPath": "params_shard_138.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.4.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "4c53024a86cbfe4710305796c5cbd9b0" }, { "dataPath": "params_shard_139.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.4.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "0103920dd4afc04bfa6314996e65ddde" }, { "dataPath": "params_shard_140.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.4.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "1d108dbb38e62f04ddbd008c3c378a3b" }, { "dataPath": "params_shard_141.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.39.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.39.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.4.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.4.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.4.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.4.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.4.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.4.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "98d5141fc825ae5df2b8647c98739bbb" }, { "dataPath": "params_shard_142.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.40.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "57ce004aa53efb9683b77fcced2fa9db" }, { "dataPath": "params_shard_143.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.40.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "c7d6707ebeab065946d541436b273a78" }, { "dataPath": "params_shard_144.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.40.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "d6295abd07bf851cb2091f6ff7fc0f8d" }, { "dataPath": "params_shard_145.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.4.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.4.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.40.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.40.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.40.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.40.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.40.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.40.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "514c49786bc874beaec6a42b0a55241f" }, { "dataPath": "params_shard_146.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.41.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "4dfa7bb206bdf06d743d4653fed8d554" }, { "dataPath": "params_shard_147.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.41.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "39bb81753baf0fbf19dd97c113e3e6f7" }, { "dataPath": "params_shard_148.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.41.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "03db929bfd9b7668bd05603279a4fd64" }, { "dataPath": "params_shard_149.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.40.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.40.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.41.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.41.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.41.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.41.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.41.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.41.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "6c3da7f9e528d48a46f8b6329193ca29" }, { "dataPath": "params_shard_150.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.42.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "2b240ead83975fbd2d476cacfebf7628" }, { "dataPath": "params_shard_151.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.42.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "603f3f0a53ee5a1e1dfdbc1e750bff8c" }, { "dataPath": "params_shard_152.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.42.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "7098f5f846fc03b982f7a2f819d13f58" }, { "dataPath": "params_shard_153.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.41.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.41.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.42.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.42.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.42.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.42.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.42.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.42.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "2e145166e749ed3a2deb2d98d4f2ea4c" }, { "dataPath": "params_shard_154.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.43.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "54c6cd2f5bb63f5f2e1d8a42407b9e9d" }, { "dataPath": "params_shard_155.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.43.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "26972dcd32ce7db7491fb5dbef2e2211" }, { "dataPath": "params_shard_156.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.43.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "749db8b0096191defc862c890dcdd69c" }, { "dataPath": "params_shard_157.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.42.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.42.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.43.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.43.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.43.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.43.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.43.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.43.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "ec61a416225ea0dcf19db0972a7c7997" }, { "dataPath": "params_shard_158.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.44.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "2c88380afa6c4b8e3b93b1fcdc811478" }, { "dataPath": "params_shard_159.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.44.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "b3e5cc734e4e49debd3ed63dc4169d8e" }, { "dataPath": "params_shard_160.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.44.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "78970a5d08988b225c24ef996ce51c06" }, { "dataPath": "params_shard_161.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.43.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.43.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.44.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.44.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.44.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.44.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.44.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.44.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "ab1d7cfba0d132f5845160ff43843e3d" }, { "dataPath": "params_shard_162.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.45.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "a85c3f9f4e40c637a0ec5c354bfb517c" }, { "dataPath": "params_shard_163.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.45.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "9da88ad77e13546c2c38021af17f848e" }, { "dataPath": "params_shard_164.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.45.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "b04b22312cdcd450fa4e9cf9f6ba4048" }, { "dataPath": "params_shard_165.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.44.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.44.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.45.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.45.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.45.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.45.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.45.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.45.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "eb62453d3dff90d2a45c3b22bfac2757" }, { "dataPath": "params_shard_166.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.46.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "13e67d6048e3d5f0e9bc117e69ac6bd3" }, { "dataPath": "params_shard_167.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.46.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "bb57c44dedf2df2f09cebfd9d23b03d2" }, { "dataPath": "params_shard_168.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.46.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "e25162fbc88f7836f0b22f2786d40513" }, { "dataPath": "params_shard_169.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.45.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.45.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.46.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.46.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.46.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.46.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.46.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.46.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "233fc4d534a9da9b0e97d820ec9fa598" }, { "dataPath": "params_shard_170.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.47.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "931e0be168bafa7786c7b3b960658674" }, { "dataPath": "params_shard_171.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.47.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "813e50b832de0875c33d5eb31f035439" }, { "dataPath": "params_shard_172.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.47.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "c820fd7cc4d01ef6069c0cca3690ce84" }, { "dataPath": "params_shard_173.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.46.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.46.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.47.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.47.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.47.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.47.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.47.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.47.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "e53813a16e9f8ddc088a5c86b85c1306" }, { "dataPath": "params_shard_174.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.48.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "8cfc852c595af8f2f2d7cf3186d1f178" }, { "dataPath": "params_shard_175.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.48.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "b7634fc36d7f679b8cffe1290129fc32" }, { "dataPath": "params_shard_176.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.48.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "37b13cab70f3d2426cfc7669c2a3c7fc" }, { "dataPath": "params_shard_177.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.47.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.47.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.48.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.48.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.48.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.48.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.48.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.48.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "0edf5eb90a550d07255ddccbffbaaae7" }, { "dataPath": "params_shard_178.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.49.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "5721921e13f5315498cf44e4c0b25767" }, { "dataPath": "params_shard_179.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.49.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "1d737df65bf603ebfe8b90724542f1cb" }, { "dataPath": "params_shard_180.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.49.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "4390e0ec530a230ca179346c769cf9b7" }, { "dataPath": "params_shard_181.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.48.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.48.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.49.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.49.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.49.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.49.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.49.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.49.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "ee465ef35a32b0554abe30bd13d9f713" }, { "dataPath": "params_shard_182.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.5.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "db5af51aca3ba6fe7f9b4e15fe89b317" }, { "dataPath": "params_shard_183.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.5.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "1601b252dea601b4d315a5d3ecd1159c" }, { "dataPath": "params_shard_184.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.5.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "fdd7cf9218d30348cf9c1e69d6294860" }, { "dataPath": "params_shard_185.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.49.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.49.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.5.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.5.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.5.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.5.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.5.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.5.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "774648aeeb77955054204fa38e59a0f4" }, { "dataPath": "params_shard_186.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.50.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "e8e5f3158199a488fa21c166103b4a4c" }, { "dataPath": "params_shard_187.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.50.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "98df265502ee4f06daeed9fff183b3c0" }, { "dataPath": "params_shard_188.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.50.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "9701572a20bad0a26a68ce69ae2853cb" }, { "dataPath": "params_shard_189.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.5.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.5.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.50.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.50.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.50.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.50.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.50.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.50.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "cea943609bbe6f16733b81d223704cae" }, { "dataPath": "params_shard_190.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.51.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "559ec506fd962b8c1b4334e3e13818c8" }, { "dataPath": "params_shard_191.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.51.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "3109ee08f2472a0d50232dd21d392129" }, { "dataPath": "params_shard_192.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.51.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "d714085c49fa623aa6e1b16a9f8ed885" }, { "dataPath": "params_shard_193.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.50.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.50.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.51.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.51.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.51.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.51.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.51.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.51.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "aef42a9ec7f6e7a37c2da5027f358d42" }, { "dataPath": "params_shard_194.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.52.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "d8fc5673b0977fced196ada05db4b304" }, { "dataPath": "params_shard_195.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.52.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "8802d8ff167987eade330a9b6887748e" }, { "dataPath": "params_shard_196.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.52.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "e707bf3a94282fab7f4c1478db10bf1a" }, { "dataPath": "params_shard_197.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.51.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.51.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.52.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.52.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.52.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.52.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.52.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.52.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "890758016a884b20dfb1dfa75480c5d1" }, { "dataPath": "params_shard_198.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.53.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "954f88e9af003038aeaf89b0a50dff5f" }, { "dataPath": "params_shard_199.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.53.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "11d0f9c5f9e389397ffec8f9c612785f" }, { "dataPath": "params_shard_200.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.53.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "aba9595caf90f417c7de747076515d98" }, { "dataPath": "params_shard_201.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.52.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.52.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.53.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.53.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.53.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.53.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.53.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.53.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "66806e6e813879b44bfd8cb8e9da1634" }, { "dataPath": "params_shard_202.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.54.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "49582c7d608bcd4495064da2e4dd4792" }, { "dataPath": "params_shard_203.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.54.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "68771d65a4e269f175f189f6616e109f" }, { "dataPath": "params_shard_204.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.54.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "5ee2586c674f4a2fe47383b1bcb74aea" }, { "dataPath": "params_shard_205.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.53.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.53.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.54.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.54.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.54.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.54.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.54.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.54.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "0a3496a1365fcdcefeb2cea442bdcfe6" }, { "dataPath": "params_shard_206.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.55.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "12f86394d8bd02f52d35d9aa95f493dd" }, { "dataPath": "params_shard_207.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.55.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "254319f0efb22422b8071a97d91e2ecb" }, { "dataPath": "params_shard_208.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.55.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "aa6b4443e397d0563c2ece4c80423ecb" }, { "dataPath": "params_shard_209.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.54.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.54.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.55.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.55.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.55.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.55.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.55.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.55.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "a7876285ac739a0d8112506241f0928f" }, { "dataPath": "params_shard_210.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.56.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "8bed260f228b8d8815b2d4a07df83249" }, { "dataPath": "params_shard_211.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.56.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "f6d3bf269a3d3ad5d1243e35ba97edd2" }, { "dataPath": "params_shard_212.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.56.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "4a0675d6af9a1b75d805cca0b4e9d778" }, { "dataPath": "params_shard_213.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.55.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.55.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.56.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.56.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.56.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.56.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.56.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.56.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "8eb4bf03189225d7eac44b545073e323" }, { "dataPath": "params_shard_214.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.57.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "958ec0a10a1238ceba7c8f337fad4cca" }, { "dataPath": "params_shard_215.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.57.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "1f0e767a99c722bacdb4f95a31ebc675" }, { "dataPath": "params_shard_216.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.57.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "fc481c9875755100b0109ecf56b3d6ef" }, { "dataPath": "params_shard_217.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.56.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.56.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.57.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.57.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.57.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.57.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.57.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.57.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "376d282e6589fc03ad6390feb284103e" }, { "dataPath": "params_shard_218.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.58.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "a92491223f82cff20da2991c809594d4" }, { "dataPath": "params_shard_219.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.58.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "a9063a67c94b155b34da3b03fee6a0e1" }, { "dataPath": "params_shard_220.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.58.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "222832211667ba4a514d6c938bf61de5" }, { "dataPath": "params_shard_221.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.57.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.57.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.58.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.58.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.58.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.58.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.58.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.58.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "7ec68ed45e228f1bd119459e173158b1" }, { "dataPath": "params_shard_222.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.59.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "900272ba33f0dbe0cb795011196ebc98" }, { "dataPath": "params_shard_223.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.59.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "0ddea0e1b595d7f05f5b9f48ca5e5927" }, { "dataPath": "params_shard_224.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.59.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "2fb77a76c1c70317bc432894d304509b" }, { "dataPath": "params_shard_225.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.58.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.58.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.59.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.59.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.59.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.59.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.59.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.59.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "becd6743287e93fdad9fcb66ae920cc4" }, { "dataPath": "params_shard_226.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.6.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "3bc907f0fa0d249416f62adcc7ba5570" }, { "dataPath": "params_shard_227.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.6.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "44e24be96e097933545b6de3607849eb" }, { "dataPath": "params_shard_228.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.6.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "f54702ac3957bdfc071ae704bfda0ca3" }, { "dataPath": "params_shard_229.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.59.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.59.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.6.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.6.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.6.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.6.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.6.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.6.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "e749f922ef3d42ae3ef573fc43b4bf95" }, { "dataPath": "params_shard_230.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.60.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "75606d43afac03cc708882f491001495" }, { "dataPath": "params_shard_231.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.60.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "8054db8b719476084789262a2aae932e" }, { "dataPath": "params_shard_232.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.60.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "4512381c6da51a4f80d8fc51dcef5abe" }, { "dataPath": "params_shard_233.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.6.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.6.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.60.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.60.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.60.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.60.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.60.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.60.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "835a50848ceabd842ad4e980ff151962" }, { "dataPath": "params_shard_234.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.61.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "b66c4d2c0c06ca1d7a8e4cfdc9a85de2" }, { "dataPath": "params_shard_235.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.61.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "2bcb09643700bd748ee05cb33fc731e7" }, { "dataPath": "params_shard_236.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.61.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "ad77a5475164b41cbfcb0e81eb0b1acd" }, { "dataPath": "params_shard_237.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.60.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.60.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.61.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.61.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.61.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.61.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.61.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.61.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "5e8af5dc367fb6ec1d5e8248e35349a8" }, { "dataPath": "params_shard_238.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.62.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "8c3820194a957199f0e1e644c8375ed4" }, { "dataPath": "params_shard_239.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.62.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "f51c68a95d12055604574f82a0795e69" }, { "dataPath": "params_shard_240.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.62.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "4471a3da0e2f3b0473e79c5645fb20da" }, { "dataPath": "params_shard_241.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.61.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.61.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.62.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.62.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.62.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.62.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.62.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.62.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "48e2272e5ded5888d348726da5b4c95f" }, { "dataPath": "params_shard_242.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.63.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "8e6553ad8cb5698d0f4e4b313e2a4055" }, { "dataPath": "params_shard_243.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.63.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "c841ef471d7233af72916e5fc5e77dee" }, { "dataPath": "params_shard_244.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.63.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "db87daaa7b47476fc08f1ff9c3245413" }, { "dataPath": "params_shard_245.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.62.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.62.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.63.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.63.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.63.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.63.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.63.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.63.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "ed70eab8bda12d145d3e5f15c7e080f4" }, { "dataPath": "params_shard_246.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.7.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "e214466d02e4165f0b223f495cba24e8" }, { "dataPath": "params_shard_247.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.7.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "48cda8b752e8b52b4817d7084ada49ee" }, { "dataPath": "params_shard_248.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.7.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "11ae3450a6e4f746ab6001c9c48934d4" }, { "dataPath": "params_shard_249.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.63.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.63.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.7.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.7.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.7.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.7.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.7.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.7.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "7727984aeb38cdb4a370a605e21308e1" }, { "dataPath": "params_shard_250.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.8.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "4428e853e7c438cec73111e634cb6caa" }, { "dataPath": "params_shard_251.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.8.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "47e0f3a906a81275b49628ee97aade5c" }, { "dataPath": "params_shard_252.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.8.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "c7f8a6fa26eb6a60955e78d912d78450" }, { "dataPath": "params_shard_253.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.7.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.7.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.8.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.8.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.8.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.8.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.8.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.8.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "e352232bdfa305486591b5f28497273f" }, { "dataPath": "params_shard_254.bin", "format": "raw-shard", "nbytes": 70778880, "records": [ { "name": "model.layers.9.mlp.down_proj.q_weight", "shape": [ 3456, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 70778880, "byteOffset": 0 } ], "md5sum": "054cea9e004788a2439269435b109938" }, { "dataPath": "params_shard_255.bin", "format": "raw-shard", "nbytes": 141557760, "records": [ { "name": "model.layers.9.mlp.gate_up_proj.q_weight", "shape": [ 640, 55296 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 141557760, "byteOffset": 0 } ], "md5sum": "b6845979db7cb3e7bc4437edf4124aad" }, { "dataPath": "params_shard_256.bin", "format": "raw-shard", "nbytes": 18350080, "records": [ { "name": "model.layers.9.self_attn.c_attn.q_weight", "shape": [ 640, 7168 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 18350080, "byteOffset": 0 } ], "md5sum": "13b5d8f627322020186425ccc2aa478e" }, { "dataPath": "params_shard_257.bin", "format": "raw-shard", "nbytes": 20760576, "records": [ { "name": "model.layers.8.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.8.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.layers.9.input_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 }, { "name": "model.layers.9.mlp.down_proj.q_scale", "shape": [ 216, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 2211840, "byteOffset": 13527040 }, { "name": "model.layers.9.mlp.gate_up_proj.q_scale", "shape": [ 40, 55296 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4423680, "byteOffset": 15738880 }, { "name": "model.layers.9.post_attention_layernorm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 20162560 }, { "name": "model.layers.9.self_attn.c_attn.bias", "shape": [ 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 14336, "byteOffset": 20172800 }, { "name": "model.layers.9.self_attn.c_attn.q_scale", "shape": [ 40, 7168 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 573440, "byteOffset": 20187136 } ], "md5sum": "a8bcd75c6427dfb67dbc0dd2391852dc" }, { "dataPath": "params_shard_258.bin", "format": "raw-shard", "nbytes": 13527040, "records": [ { "name": "model.layers.9.self_attn.o_proj.q_weight", "shape": [ 640, 5120 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "model.layers.9.self_attn.o_proj.q_scale", "shape": [ 40, 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 13107200 }, { "name": "model.norm.weight", "shape": [ 5120 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 10240, "byteOffset": 13516800 } ], "md5sum": "1b02d0515adfbc5cc336f7be37e0f97e" } ] }