{ "metadata": { "ParamSize": 283, "ParamBytes": 1807423488.0, "BitsPerParam": 4.500626782697164 }, "records": [ { "dataPath": "params_shard_0.bin", "format": "raw-shard", "nbytes": 197001216, "records": [ { "name": "model.embed_tokens.q_weight", "shape": [ 128256, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 197001216, "byteOffset": 0 } ], "md5sum": "1e0b525f7fc8385a67d88c57159c217d" }, { "dataPath": "params_shard_1.bin", "format": "raw-shard", "nbytes": 24631296, "records": [ { "name": "model.embed_tokens.q_scale", "shape": [ 128256, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 24625152, "byteOffset": 0 }, { "name": "model.layers.0.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 24625152 } ], "md5sum": "456d503975d868cec3199de1048e6e9d" }, { "dataPath": "params_shard_2.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.0.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "32d771d88d9bc076880262b4395051dc" }, { "dataPath": "params_shard_3.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.0.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.0.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.0.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.0.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.0.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.0.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.0.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.0.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.1.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "475a75f73f5535f2b1d3e04b7c9fff1c" }, { "dataPath": "params_shard_4.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.1.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "fdf6c93aec8167caa07379ef0af37cfd" }, { "dataPath": "params_shard_5.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.1.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.1.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.1.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.1.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.1.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.1.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.1.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.1.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.2.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "ef635cdf24cfec91e0867b181e4add65" }, { "dataPath": "params_shard_6.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.2.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "58030e50a8ff0edb4e70aea0a2452475" }, { "dataPath": "params_shard_7.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.2.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.2.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.2.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.2.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.2.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.2.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.2.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.2.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.3.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "e5f4ee2fcfd2300db441757e0df08d44" }, { "dataPath": "params_shard_8.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.3.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "ce0f4ad852d29cc82e114eaad79535f1" }, { "dataPath": "params_shard_9.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.3.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.3.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.3.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.3.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.3.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.3.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.3.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.3.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.4.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "7a21a1031382848490d095bfe165ebdc" }, { "dataPath": "params_shard_10.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.4.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "f6f5f1a300f993b974d514a55290d7a3" }, { "dataPath": "params_shard_11.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.4.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.4.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.4.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.4.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.4.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.4.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.4.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.4.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.5.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "d24c86b075c790908accdef01b8d4f9d" }, { "dataPath": "params_shard_12.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.5.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "d6415d678108238d3bdd238d71106949" }, { "dataPath": "params_shard_13.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.5.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.5.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.5.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.5.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.5.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.5.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.5.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.5.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.6.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "c5b141ee8a2562d71127c124f8c6012c" }, { "dataPath": "params_shard_14.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.6.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "039352e548b48eba022004db2e53ff48" }, { "dataPath": "params_shard_15.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.6.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.6.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.6.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.6.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.6.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.6.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.6.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.6.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.7.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "93cc8392182640d2affd3026c5e1d1af" }, { "dataPath": "params_shard_16.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.7.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "9fb4ca7b86843191bd02194f84832033" }, { "dataPath": "params_shard_17.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.8.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "19ab539d846dc9b1732679807290c56a" }, { "dataPath": "params_shard_18.bin", "format": "raw-shard", "nbytes": 31463424, "records": [ { "name": "model.layers.7.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.7.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.7.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.7.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.7.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.7.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.7.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.7.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 } ], "md5sum": "71c76f43ddeb3cd488291ac37a7676f6" }, { "dataPath": "params_shard_19.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.10.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "c45c97a699c667fc957f4ac2f0bc9b42" }, { "dataPath": "params_shard_20.bin", "format": "raw-shard", "nbytes": 31463424, "records": [ { "name": "model.layers.8.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 0 }, { "name": "model.layers.8.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 3145728 }, { "name": "model.layers.8.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 11010048 }, { "name": "model.layers.8.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 11993088 }, { "name": "model.layers.8.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 16711680 }, { "name": "model.layers.10.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.10.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 17307648 }, { "name": "model.layers.10.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 29890560 } ], "md5sum": "a604c825bbca9b8c768fdfd416c27ea3" }, { "dataPath": "params_shard_21.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.11.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "77a8fccf48b173ead28ab61017fa127e" }, { "dataPath": "params_shard_22.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.10.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 0 }, { "name": "model.layers.10.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 3145728 }, { "name": "model.layers.10.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 3151872 }, { "name": "model.layers.10.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 11016192 }, { "name": "model.layers.10.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 11999232 }, { "name": "model.layers.10.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 16717824 }, { "name": "model.layers.11.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17307648 }, { "name": "model.layers.11.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 17313792 }, { "name": "model.layers.11.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 29896704 } ], "md5sum": "7a7f4651c2f6d6f2217ca3ed51646601" }, { "dataPath": "params_shard_23.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.12.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "246ef7684dc97af1410b6e1ac10f34ab" }, { "dataPath": "params_shard_24.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.11.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 0 }, { "name": "model.layers.11.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 3145728 }, { "name": "model.layers.11.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 3151872 }, { "name": "model.layers.11.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 11016192 }, { "name": "model.layers.11.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 11999232 }, { "name": "model.layers.11.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 16717824 }, { "name": "model.layers.12.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17307648 }, { "name": "model.layers.12.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 17313792 }, { "name": "model.layers.12.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 29896704 } ], "md5sum": "1cecfa222ef7daba194f7ac217121840" }, { "dataPath": "params_shard_25.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.13.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "dff1898bb46976470255ce92526820f8" }, { "dataPath": "params_shard_26.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.12.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 0 }, { "name": "model.layers.12.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 3145728 }, { "name": "model.layers.12.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 3151872 }, { "name": "model.layers.12.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 11016192 }, { "name": "model.layers.12.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 11999232 }, { "name": "model.layers.12.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 16717824 }, { "name": "model.layers.13.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17307648 }, { "name": "model.layers.13.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 17313792 }, { "name": "model.layers.13.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 29896704 } ], "md5sum": "7f0ce08ef0c00349b2eeaa9412a66117" }, { "dataPath": "params_shard_27.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.14.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "3ffec6b1de305ad0ab3a8b9326802305" }, { "dataPath": "params_shard_28.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.13.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 0 }, { "name": "model.layers.13.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 3145728 }, { "name": "model.layers.13.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 3151872 }, { "name": "model.layers.13.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 11016192 }, { "name": "model.layers.13.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 11999232 }, { "name": "model.layers.13.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 16717824 }, { "name": "model.layers.14.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17307648 }, { "name": "model.layers.14.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 17313792 }, { "name": "model.layers.14.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 29896704 } ], "md5sum": "f7f31df9b50a2fbfb0db0cdb5ccf2bb6" }, { "dataPath": "params_shard_29.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.15.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "2ff80a319216d7f3933a57c5496efb68" }, { "dataPath": "params_shard_30.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.14.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 0 }, { "name": "model.layers.14.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 3145728 }, { "name": "model.layers.14.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 3151872 }, { "name": "model.layers.14.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 11016192 }, { "name": "model.layers.14.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 11999232 }, { "name": "model.layers.14.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 16717824 }, { "name": "model.layers.15.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17307648 }, { "name": "model.layers.15.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 17313792 }, { "name": "model.layers.15.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 29896704 } ], "md5sum": "70435f6c1cd32da5a826b33e874ea061" }, { "dataPath": "params_shard_31.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.16.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "0ded49d082c3dcca5271cde9402b792c" }, { "dataPath": "params_shard_32.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.15.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 0 }, { "name": "model.layers.15.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 3145728 }, { "name": "model.layers.15.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 3151872 }, { "name": "model.layers.15.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 11016192 }, { "name": "model.layers.15.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 11999232 }, { "name": "model.layers.15.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 16717824 }, { "name": "model.layers.16.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17307648 }, { "name": "model.layers.16.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 17313792 }, { "name": "model.layers.16.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 29896704 } ], "md5sum": "a8f60ed23ed56c046f6219d712350159" }, { "dataPath": "params_shard_33.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.17.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "52ece1fa52c3b0c9372bd06ee451011b" }, { "dataPath": "params_shard_34.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.16.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 0 }, { "name": "model.layers.16.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 3145728 }, { "name": "model.layers.16.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 3151872 }, { "name": "model.layers.16.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 11016192 }, { "name": "model.layers.16.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 11999232 }, { "name": "model.layers.16.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 16717824 }, { "name": "model.layers.17.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17307648 }, { "name": "model.layers.17.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 17313792 }, { "name": "model.layers.17.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 29896704 } ], "md5sum": "3d67c8f1502fc71770f412cdcfcae30e" }, { "dataPath": "params_shard_35.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.18.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "997d526f8e77e533bfa110ffed83c3f9" }, { "dataPath": "params_shard_36.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.17.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 0 }, { "name": "model.layers.17.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 3145728 }, { "name": "model.layers.17.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 3151872 }, { "name": "model.layers.17.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 11016192 }, { "name": "model.layers.17.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 11999232 }, { "name": "model.layers.17.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 16717824 }, { "name": "model.layers.18.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17307648 }, { "name": "model.layers.18.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 17313792 }, { "name": "model.layers.18.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 29896704 } ], "md5sum": "fd0fbe44728aa686bbdd048ba62b0469" }, { "dataPath": "params_shard_37.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.19.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "c1ccab89779c20e7d1dcdda407f9b7e8" }, { "dataPath": "params_shard_38.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.18.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 0 }, { "name": "model.layers.18.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 3145728 }, { "name": "model.layers.18.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 3151872 }, { "name": "model.layers.18.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 11016192 }, { "name": "model.layers.18.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 11999232 }, { "name": "model.layers.18.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 16717824 }, { "name": "model.layers.19.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17307648 }, { "name": "model.layers.19.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 17313792 }, { "name": "model.layers.19.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 29896704 } ], "md5sum": "8e8da5c9d3c09f25c73ccfb8a4050847" }, { "dataPath": "params_shard_39.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.20.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "3ffc9f9599c0172607446c5d491252a7" }, { "dataPath": "params_shard_40.bin", "format": "raw-shard", "nbytes": 29300736, "records": [ { "name": "model.layers.19.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 0 }, { "name": "model.layers.19.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 3145728 }, { "name": "model.layers.19.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 3151872 }, { "name": "model.layers.19.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 11016192 }, { "name": "model.layers.19.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 11999232 }, { "name": "model.layers.19.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 16717824 }, { "name": "model.layers.20.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 17307648 }, { "name": "model.layers.20.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 20453376 }, { "name": "model.layers.20.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 28317696 } ], "md5sum": "95ca43293edc2a4a88d35c4397dafb85" }, { "dataPath": "params_shard_41.bin", "format": "raw-shard", "nbytes": 32065536, "records": [ { "name": "model.layers.20.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 0 }, { "name": "model.layers.20.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 4718592 }, { "name": "model.layers.8.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 5308416 }, { "name": "model.layers.8.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 5314560 }, { "name": "model.layers.8.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 17897472 }, { "name": "model.layers.8.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19470336 }, { "name": "model.layers.9.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 19476480 }, { "name": "model.layers.9.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 19482624 } ], "md5sum": "4cc8fc88925a88d521d4b34c98c912b0" }, { "dataPath": "params_shard_42.bin", "format": "raw-shard", "nbytes": 29890560, "records": [ { "name": "model.layers.9.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 0 }, { "name": "model.layers.9.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 1572864 }, { "name": "model.layers.9.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 26738688 }, { "name": "model.layers.9.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 29884416 } ], "md5sum": "643f47c5704afffd2274b57ceb0a83f3" }, { "dataPath": "params_shard_43.bin", "format": "raw-shard", "nbytes": 28329984, "records": [ { "name": "model.layers.9.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 0 }, { "name": "model.layers.9.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 7864320 }, { "name": "model.layers.9.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 8847360 }, { "name": "model.layers.9.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 13565952 }, { "name": "model.layers.20.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 14155776 }, { "name": "model.layers.20.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 14161920 }, { "name": "model.layers.20.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 26744832 }, { "name": "model.layers.20.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 28317696 }, { "name": "model.layers.21.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 28323840 } ], "md5sum": "f6b89706c99f5bd9d3fc24a22c4e6ee7" }, { "dataPath": "params_shard_44.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.21.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "d5284a8c3d6d8b02083536fccb0302a7" }, { "dataPath": "params_shard_45.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.21.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.21.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.21.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.21.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.21.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.21.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.21.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.21.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.22.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "9bda83760add6f4308887ee68956b9df" }, { "dataPath": "params_shard_46.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.22.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "fc5e9bd5cbe0ae370e37fcc8a35c10b8" }, { "dataPath": "params_shard_47.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.22.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.22.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.22.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.22.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.22.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.22.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.22.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.22.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.23.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "666d72499a7b884afc22971b4e69f96e" }, { "dataPath": "params_shard_48.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.23.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "d772d25677e85516ba220c6ed1feda20" }, { "dataPath": "params_shard_49.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.23.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.23.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.23.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.23.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.23.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.23.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.23.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.23.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.24.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "057cc3d8e4a6a67bccfec7372bae573b" }, { "dataPath": "params_shard_50.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.24.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "5384a78b004acd9e5fa50ed4bfc44360" }, { "dataPath": "params_shard_51.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.24.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.24.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.24.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.24.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.24.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.24.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.24.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.24.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.25.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "b52dc9b67dfa0fc706c888ba2ea95e28" }, { "dataPath": "params_shard_52.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.25.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "be8946267ef5112cb5c83382610fe7d2" }, { "dataPath": "params_shard_53.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.25.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.25.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.25.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.25.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.25.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.25.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.25.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.25.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.26.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "6c7c6412d5e1da6287101481e6a51832" }, { "dataPath": "params_shard_54.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.26.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "a8b26bfe054d0eafb9affa9c2f506478" }, { "dataPath": "params_shard_55.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.26.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.26.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.26.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.26.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.26.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.26.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.26.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.26.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.layers.27.input_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "9838855eefdb76d5e2dbf3677aeac9de" }, { "dataPath": "params_shard_56.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.27.mlp.gate_up_proj.q_weight", "shape": [ 16384, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "864c1d0f3196757426bf472e9b903d50" }, { "dataPath": "params_shard_57.bin", "format": "raw-shard", "nbytes": 31469568, "records": [ { "name": "model.layers.27.mlp.down_proj.q_weight", "shape": [ 3072, 1024 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 12582912, "byteOffset": 0 }, { "name": "model.layers.27.mlp.down_proj.q_scale", "shape": [ 3072, 256 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1572864, "byteOffset": 12582912 }, { "name": "model.layers.27.mlp.gate_up_proj.q_scale", "shape": [ 16384, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3145728, "byteOffset": 14155776 }, { "name": "model.layers.27.post_attention_layernorm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 17301504 }, { "name": "model.layers.27.self_attn.qkv_proj.q_weight", "shape": [ 5120, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 17307648 }, { "name": "model.layers.27.self_attn.qkv_proj.q_scale", "shape": [ 5120, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 983040, "byteOffset": 25171968 }, { "name": "model.layers.27.self_attn.o_proj.q_weight", "shape": [ 3072, 384 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 26155008 }, { "name": "model.layers.27.self_attn.o_proj.q_scale", "shape": [ 3072, 96 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 589824, "byteOffset": 30873600 }, { "name": "model.norm.weight", "shape": [ 3072 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6144, "byteOffset": 31463424 } ], "md5sum": "9b03583c425405503d0f68da737c06ba" } ] }