{ "metadata": { "ParamSize": 199, "ParamBytes": 3554176000.0, "BitsPerParam": 16.0 }, "records": [ { "dataPath": "params_shard_0.bin", "format": "raw-shard", "nbytes": 466747392, "records": [ { "name": "lm_head.weight", "shape": [ 151936, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 466747392, "byteOffset": 0 } ], "md5sum": "4d8a904b17c498e9785aa867e17d7c62" }, { "dataPath": "params_shard_1.bin", "format": "raw-shard", "nbytes": 466747392, "records": [ { "name": "model.embed_tokens.weight", "shape": [ 151936, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 466747392, "byteOffset": 0 } ], "md5sum": "22e897f1f35081a84326de4fde176669" }, { "dataPath": "params_shard_2.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.0.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "81995238ec09105c42af2ff9cf79fc81" }, { "dataPath": "params_shard_3.bin", "format": "raw-shard", "nbytes": 27535360, "records": [ { "name": "model.layers.0.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 0 }, { "name": "model.layers.0.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 3072 }, { "name": "model.layers.0.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 27528192 }, { "name": "model.layers.0.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 27531264 } ], "md5sum": "1a2e1e13b84dfdee8d5957a6ecb4dd7c" }, { "dataPath": "params_shard_4.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.1.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "9decd2b72296182323d68b6233a3e47a" }, { "dataPath": "params_shard_5.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.1.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "46f2860673f77add88bbe1e6d697be1b" }, { "dataPath": "params_shard_6.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.10.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "f260e42af4a1bc32aba36b7fd7d31116" }, { "dataPath": "params_shard_7.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.10.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "2d3c8efd9e683084ef3bdc39ca2d65da" }, { "dataPath": "params_shard_8.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.11.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "4aebcfa9094ddc7eb84c1a0f7b8edcd5" }, { "dataPath": "params_shard_9.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.11.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "28cd958ad4a9d0605066a9e1f2e14c0d" }, { "dataPath": "params_shard_10.bin", "format": "raw-shard", "nbytes": 33060864, "records": [ { "name": "model.layers.0.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 0 }, { "name": "model.layers.0.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 6291456 }, { "name": "model.layers.1.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11010048 }, { "name": "model.layers.1.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11013120 }, { "name": "model.layers.1.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 11016192 }, { "name": "model.layers.1.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 11020288 }, { "name": "model.layers.1.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17311744 }, { "name": "model.layers.10.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22030336 }, { "name": "model.layers.10.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22033408 }, { "name": "model.layers.10.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22036480 }, { "name": "model.layers.10.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 22040576 }, { "name": "model.layers.10.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 28332032 }, { "name": "model.layers.11.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33050624 }, { "name": "model.layers.11.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33053696 }, { "name": "model.layers.11.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 33056768 } ], "md5sum": "724eafae7080ac9c64e2635b8f39dee1" }, { "dataPath": "params_shard_11.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.12.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "7c8fbc4469b959ca36ccb5b56b1aff7a" }, { "dataPath": "params_shard_12.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.12.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "7dcc2b209c32558cf604f372054f94e1" }, { "dataPath": "params_shard_13.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.13.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "6c6d9014c54fc4be4c62c4d76bded576" }, { "dataPath": "params_shard_14.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.13.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "783503713029d2e2e1e5ef970dbb4659" }, { "dataPath": "params_shard_15.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.14.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "3c3ad6b68589b515ff032534668446d7" }, { "dataPath": "params_shard_16.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.14.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "829ef33937e95753fa00e48e71d26079" }, { "dataPath": "params_shard_17.bin", "format": "raw-shard", "nbytes": 33060864, "records": [ { "name": "model.layers.11.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 0 }, { "name": "model.layers.11.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 6291456 }, { "name": "model.layers.12.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11010048 }, { "name": "model.layers.12.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11013120 }, { "name": "model.layers.12.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 11016192 }, { "name": "model.layers.12.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 11020288 }, { "name": "model.layers.12.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17311744 }, { "name": "model.layers.13.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22030336 }, { "name": "model.layers.13.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22033408 }, { "name": "model.layers.13.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22036480 }, { "name": "model.layers.13.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 22040576 }, { "name": "model.layers.13.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 28332032 }, { "name": "model.layers.14.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33050624 }, { "name": "model.layers.14.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33053696 }, { "name": "model.layers.14.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 33056768 } ], "md5sum": "1011c7025b617580675ce288695c9320" }, { "dataPath": "params_shard_18.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.15.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "1ec33da19abd8946ec3ce6eff05a00f2" }, { "dataPath": "params_shard_19.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.15.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "80bada2591e989c2a2d893ac7ca17d92" }, { "dataPath": "params_shard_20.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.16.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "bc2cfd054525906601b86dfaef89f799" }, { "dataPath": "params_shard_21.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.16.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "689386905f4503dc078865cead2e2516" }, { "dataPath": "params_shard_22.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.17.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "9f074861b1282a53d624aa8271eb7ceb" }, { "dataPath": "params_shard_23.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.17.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "2151c0b55962940f406023b7abb50ced" }, { "dataPath": "params_shard_24.bin", "format": "raw-shard", "nbytes": 33060864, "records": [ { "name": "model.layers.14.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 0 }, { "name": "model.layers.14.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 6291456 }, { "name": "model.layers.15.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11010048 }, { "name": "model.layers.15.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11013120 }, { "name": "model.layers.15.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 11016192 }, { "name": "model.layers.15.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 11020288 }, { "name": "model.layers.15.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17311744 }, { "name": "model.layers.16.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22030336 }, { "name": "model.layers.16.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22033408 }, { "name": "model.layers.16.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22036480 }, { "name": "model.layers.16.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 22040576 }, { "name": "model.layers.16.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 28332032 }, { "name": "model.layers.17.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33050624 }, { "name": "model.layers.17.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33053696 }, { "name": "model.layers.17.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 33056768 } ], "md5sum": "e0f33779f14209c649609770249e910e" }, { "dataPath": "params_shard_25.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.18.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "6be106e14a60e9192d92b9bc5e03ab63" }, { "dataPath": "params_shard_26.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.18.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "730cee18923347893bbb44b3653dcd4c" }, { "dataPath": "params_shard_27.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.19.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "68ac956faf4373deffce9e701b13312e" }, { "dataPath": "params_shard_28.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.19.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "13cfe0b30478bdb5907d2f97062b96a6" }, { "dataPath": "params_shard_29.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.2.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "d22b27cf62cf897d4afc59b56df26dfd" }, { "dataPath": "params_shard_30.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.2.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "576f048acbf3d1be4814d7d5c8924ccf" }, { "dataPath": "params_shard_31.bin", "format": "raw-shard", "nbytes": 33060864, "records": [ { "name": "model.layers.17.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 0 }, { "name": "model.layers.17.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 6291456 }, { "name": "model.layers.18.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11010048 }, { "name": "model.layers.18.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11013120 }, { "name": "model.layers.18.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 11016192 }, { "name": "model.layers.18.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 11020288 }, { "name": "model.layers.18.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17311744 }, { "name": "model.layers.19.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22030336 }, { "name": "model.layers.19.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22033408 }, { "name": "model.layers.19.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22036480 }, { "name": "model.layers.19.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 22040576 }, { "name": "model.layers.19.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 28332032 }, { "name": "model.layers.2.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33050624 }, { "name": "model.layers.2.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33053696 }, { "name": "model.layers.2.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 33056768 } ], "md5sum": "ba35ae2d08939bf3a86b5634f1c2fc50" }, { "dataPath": "params_shard_32.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.20.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "cea4bfe4d20533262bbfd1a5f55414a0" }, { "dataPath": "params_shard_33.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.20.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "3bb671f6852a83ace483b8774e4702c1" }, { "dataPath": "params_shard_34.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.21.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "a1454ea96781f0988c17c5103285e57e" }, { "dataPath": "params_shard_35.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.21.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "8a90e1a3f9a3205d686662193403b96f" }, { "dataPath": "params_shard_36.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.22.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "d5ccdd836cd755e7b75ca6c57d52349f" }, { "dataPath": "params_shard_37.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.22.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "4c16ce0ea740f7c4cfe53c2d767c0932" }, { "dataPath": "params_shard_38.bin", "format": "raw-shard", "nbytes": 33060864, "records": [ { "name": "model.layers.2.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 0 }, { "name": "model.layers.2.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 6291456 }, { "name": "model.layers.20.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11010048 }, { "name": "model.layers.20.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11013120 }, { "name": "model.layers.20.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 11016192 }, { "name": "model.layers.20.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 11020288 }, { "name": "model.layers.20.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17311744 }, { "name": "model.layers.21.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22030336 }, { "name": "model.layers.21.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22033408 }, { "name": "model.layers.21.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22036480 }, { "name": "model.layers.21.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 22040576 }, { "name": "model.layers.21.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 28332032 }, { "name": "model.layers.22.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33050624 }, { "name": "model.layers.22.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33053696 }, { "name": "model.layers.22.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 33056768 } ], "md5sum": "f46d7d3c3f1f8ee51a5091a056e7469f" }, { "dataPath": "params_shard_39.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.23.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "67ba46d9daf2ea224cdce3df4bbee33e" }, { "dataPath": "params_shard_40.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.23.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "c8b688c31f6987ccf6fe43b600ec578c" }, { "dataPath": "params_shard_41.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.24.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "43865758cc6d7f7ce6cca424e1294f6a" }, { "dataPath": "params_shard_42.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.24.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "b344e1c3df1d97f91fac114b322f6d8a" }, { "dataPath": "params_shard_43.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.25.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "d257c98f83e522584fe43139c4e54356" }, { "dataPath": "params_shard_44.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.25.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "c2979b3dcaf44c5ab44ce23603a59710" }, { "dataPath": "params_shard_45.bin", "format": "raw-shard", "nbytes": 33060864, "records": [ { "name": "model.layers.22.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 0 }, { "name": "model.layers.22.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 6291456 }, { "name": "model.layers.23.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11010048 }, { "name": "model.layers.23.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11013120 }, { "name": "model.layers.23.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 11016192 }, { "name": "model.layers.23.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 11020288 }, { "name": "model.layers.23.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17311744 }, { "name": "model.layers.24.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22030336 }, { "name": "model.layers.24.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22033408 }, { "name": "model.layers.24.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22036480 }, { "name": "model.layers.24.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 22040576 }, { "name": "model.layers.24.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 28332032 }, { "name": "model.layers.25.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33050624 }, { "name": "model.layers.25.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33053696 }, { "name": "model.layers.25.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 33056768 } ], "md5sum": "dc33429dc1878212438734493b4de2aa" }, { "dataPath": "params_shard_46.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.26.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "a5fed699a77811991fb069ec33ec8b2f" }, { "dataPath": "params_shard_47.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.26.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "d43f6f25d946d9dab2d69d525c0fca55" }, { "dataPath": "params_shard_48.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.27.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "9104c976c3f0f79fb5ea002847b51a6a" }, { "dataPath": "params_shard_49.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.27.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "c1d597f0ccd26539aa70a0cd7b60981c" }, { "dataPath": "params_shard_50.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.3.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "010fdd102a51b95cd310a15b471487d3" }, { "dataPath": "params_shard_51.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.3.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "598d0dc39259f1ab7a9e7ace7b47f46a" }, { "dataPath": "params_shard_52.bin", "format": "raw-shard", "nbytes": 33060864, "records": [ { "name": "model.layers.25.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 0 }, { "name": "model.layers.25.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 6291456 }, { "name": "model.layers.26.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11010048 }, { "name": "model.layers.26.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11013120 }, { "name": "model.layers.26.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 11016192 }, { "name": "model.layers.26.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 11020288 }, { "name": "model.layers.26.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17311744 }, { "name": "model.layers.27.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22030336 }, { "name": "model.layers.27.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22033408 }, { "name": "model.layers.27.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22036480 }, { "name": "model.layers.27.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 22040576 }, { "name": "model.layers.27.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 28332032 }, { "name": "model.layers.3.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33050624 }, { "name": "model.layers.3.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33053696 }, { "name": "model.layers.3.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 33056768 } ], "md5sum": "f09bfdabd3e4dccc0ec537d453bd3ff0" }, { "dataPath": "params_shard_53.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.4.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "1cf8b99bb81904f0fbe87954c7c090d4" }, { "dataPath": "params_shard_54.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.4.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "62d916c6fc81913257cbd4890b0c400e" }, { "dataPath": "params_shard_55.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.5.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "a8065be5ff3a1b84fc233fa22ee79387" }, { "dataPath": "params_shard_56.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.5.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "afc16a9ee1041f8d40f4b19657550b20" }, { "dataPath": "params_shard_57.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.6.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "50356e8b1fa958c57c3d728a4955dfa1" }, { "dataPath": "params_shard_58.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.6.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "a667254802236f0f3a49a7659f14e7c9" }, { "dataPath": "params_shard_59.bin", "format": "raw-shard", "nbytes": 33060864, "records": [ { "name": "model.layers.3.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 0 }, { "name": "model.layers.3.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 6291456 }, { "name": "model.layers.4.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11010048 }, { "name": "model.layers.4.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11013120 }, { "name": "model.layers.4.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 11016192 }, { "name": "model.layers.4.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 11020288 }, { "name": "model.layers.4.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17311744 }, { "name": "model.layers.5.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22030336 }, { "name": "model.layers.5.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22033408 }, { "name": "model.layers.5.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22036480 }, { "name": "model.layers.5.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 22040576 }, { "name": "model.layers.5.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 28332032 }, { "name": "model.layers.6.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33050624 }, { "name": "model.layers.6.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33053696 }, { "name": "model.layers.6.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 33056768 } ], "md5sum": "d956f7171d618981966fc174cae6f61d" }, { "dataPath": "params_shard_60.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.7.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "645216271ed732eee9146ef4842f45a7" }, { "dataPath": "params_shard_61.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.7.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "ded72450ae4b608394f76c94faa77944" }, { "dataPath": "params_shard_62.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.8.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "b4c989a18b750af37b85a4ad69ffefbd" }, { "dataPath": "params_shard_63.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.8.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "9dbee59512c02d9c2207ca383daa2d6d" }, { "dataPath": "params_shard_64.bin", "format": "raw-shard", "nbytes": 27525120, "records": [ { "name": "model.layers.9.mlp.down_proj.weight", "shape": [ 1536, 8960 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 27525120, "byteOffset": 0 } ], "md5sum": "1b4766e9f4a576f525569989ea262855" }, { "dataPath": "params_shard_65.bin", "format": "raw-shard", "nbytes": 55050240, "records": [ { "name": "model.layers.9.mlp.gate_up_proj.weight", "shape": [ 17920, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 55050240, "byteOffset": 0 } ], "md5sum": "d9828ef62b87d02af740526feadb67e7" }, { "dataPath": "params_shard_66.bin", "format": "raw-shard", "nbytes": 33060864, "records": [ { "name": "model.layers.6.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 0 }, { "name": "model.layers.6.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 6291456 }, { "name": "model.layers.7.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11010048 }, { "name": "model.layers.7.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11013120 }, { "name": "model.layers.7.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 11016192 }, { "name": "model.layers.7.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 11020288 }, { "name": "model.layers.7.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 17311744 }, { "name": "model.layers.8.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22030336 }, { "name": "model.layers.8.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 22033408 }, { "name": "model.layers.8.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 22036480 }, { "name": "model.layers.8.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 22040576 }, { "name": "model.layers.8.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 28332032 }, { "name": "model.layers.9.input_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33050624 }, { "name": "model.layers.9.post_attention_layernorm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 33053696 }, { "name": "model.layers.9.self_attn.c_attn.bias", "shape": [ 2048 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 33056768 } ], "md5sum": "ba69a1abb6f090c015317fea9b3a3408" }, { "dataPath": "params_shard_67.bin", "format": "raw-shard", "nbytes": 11013120, "records": [ { "name": "model.layers.9.self_attn.c_attn.weight", "shape": [ 2048, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 6291456, "byteOffset": 0 }, { "name": "model.layers.9.self_attn.o_proj.weight", "shape": [ 1536, 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4718592, "byteOffset": 6291456 }, { "name": "model.norm.weight", "shape": [ 1536 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 3072, "byteOffset": 11010048 } ], "md5sum": "9b206085ecacfac239a6d4045981af0b" } ] }