{ "metadata": { "ParamSize": 220, "ParamBytes": 6578061312.0, "BitsPerParam": 32.0 }, "records": [ { "dataPath": "params_shard_0.bin", "format": "raw-shard", "nbytes": 411041792, "records": [ { "name": "lm_head.weight", "shape": [ 100352, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 411041792, "byteOffset": 0 } ], "md5sum": "8feaa03c43296af6d56d6ae34d79ff49" }, { "dataPath": "params_shard_1.bin", "format": "raw-shard", "nbytes": 411041792, "records": [ { "name": "model.embed_tokens.weight", "shape": [ 100352, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 411041792, "byteOffset": 0 } ], "md5sum": "6e70bbcc4f9a5ebec925c6ce74a85d6e" }, { "dataPath": "params_shard_2.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.0.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "1ee2c21c4c42bbfd1a96723eac48ed6f" }, { "dataPath": "params_shard_3.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.0.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "8650fd58ae045e16af2dc1f9fdbc7c60" }, { "dataPath": "params_shard_4.bin", "format": "raw-shard", "nbytes": 23068672, "records": [ { "name": "model.layers.1.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 0 } ], "md5sum": "10234dc2b68ee4e0828d1e7a8751562c" }, { "dataPath": "params_shard_5.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.1.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "fa1fc18500fca5973ad17d3129d76d1c" }, { "dataPath": "params_shard_6.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.1.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "b0aa899c673432f2ab23655f1779b287" }, { "dataPath": "params_shard_7.bin", "format": "raw-shard", "nbytes": 31514624, "records": [ { "name": "model.layers.0.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 0 }, { "name": "model.layers.0.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 4096 }, { "name": "model.layers.0.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8192 }, { "name": "model.layers.0.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23076864 }, { "name": "model.layers.0.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 23080960 }, { "name": "model.layers.0.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 23085056 }, { "name": "model.layers.0.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 23097344 }, { "name": "model.layers.1.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31485952 }, { "name": "model.layers.1.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31490048 }, { "name": "model.layers.1.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31494144 }, { "name": "model.layers.1.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31498240 }, { "name": "model.layers.1.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31502336 } ], "md5sum": "a366140336e0fc5625f067a9715bbf3f" }, { "dataPath": "params_shard_8.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.10.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "76709d8055c55eff2616cd2ee3fbb222" }, { "dataPath": "params_shard_9.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.10.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "2157fce09a8f8aaaa3762fcb9646afcf" }, { "dataPath": "params_shard_10.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.1.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.10.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.10.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.10.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.10.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.10.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.10.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "f5bb6f1e81f74526d8d921862d150950" }, { "dataPath": "params_shard_11.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.11.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "6366654f8345be2accfc828ff5be0a2f" }, { "dataPath": "params_shard_12.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.11.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "ddeaf590f32dbad9675ffb4f161aa468" }, { "dataPath": "params_shard_13.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.10.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.11.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.11.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.11.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.11.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.11.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.11.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "861caedc6a61e8de4eb1c73344b8b352" }, { "dataPath": "params_shard_14.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.12.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "54f1342270651067c9698809b8130756" }, { "dataPath": "params_shard_15.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.12.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "8ba339fa062866f94263c2f7ca7f7a6a" }, { "dataPath": "params_shard_16.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.11.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.12.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.12.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.12.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.12.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.12.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.12.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "79585b00793455bac8b7f41eb03a6531" }, { "dataPath": "params_shard_17.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.13.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "cb530c4b04e5523a06192b121beee591" }, { "dataPath": "params_shard_18.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.13.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "42561efa3873c8352ea1400b11c4eb87" }, { "dataPath": "params_shard_19.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.12.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.13.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.13.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.13.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.13.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.13.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.13.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "f270a9de3e40c9bb3289c47d3975e489" }, { "dataPath": "params_shard_20.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.14.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "96ac57d821a2148d40bf5da25776db0a" }, { "dataPath": "params_shard_21.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.14.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "535b088c9a11993bc5716e8202abba1e" }, { "dataPath": "params_shard_22.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.13.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.14.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.14.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.14.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.14.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.14.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.14.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "39bbf585ed946719daa1b63ca36ff72a" }, { "dataPath": "params_shard_23.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.15.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "cc11aba9e48fe3e0ff913800e04bbc3d" }, { "dataPath": "params_shard_24.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.15.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "b8aba12c02a13c086f49ba2c0d08f534" }, { "dataPath": "params_shard_25.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.14.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.15.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.15.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.15.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.15.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.15.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.15.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "c348dfe6bca6a8f5dd9f16a62cf76c4f" }, { "dataPath": "params_shard_26.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.16.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "627233672b35ee886132e8a36824eaa0" }, { "dataPath": "params_shard_27.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.16.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "87eac07c30d3da72dde134591c228061" }, { "dataPath": "params_shard_28.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.15.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.16.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.16.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.16.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.16.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.16.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.16.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "a30882c3d8928cc34082b5578e15e158" }, { "dataPath": "params_shard_29.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.17.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "29e98bbacba4fb105f50eb5fb6a587cf" }, { "dataPath": "params_shard_30.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.17.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "d62b14f6d9e4a43032ca3f11957aa408" }, { "dataPath": "params_shard_31.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.16.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.17.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.17.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.17.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.17.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.17.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.17.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "9042ac3354d0c7eadb8c49892880a7e2" }, { "dataPath": "params_shard_32.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.18.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "3f0701e33136482f785a5fa0393f91e5" }, { "dataPath": "params_shard_33.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.18.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "2edf3df335285fe75a275d71b0cf0e46" }, { "dataPath": "params_shard_34.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.17.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.18.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.18.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.18.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.18.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.18.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.18.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "6c50305e7eff773699910ef8c2469366" }, { "dataPath": "params_shard_35.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.19.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "4e28c8d694655bd6560df661988f48ea" }, { "dataPath": "params_shard_36.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.19.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "9bef8b9a825ce9864f1a65dff4be96ee" }, { "dataPath": "params_shard_37.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.18.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.19.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.19.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.19.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.19.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.19.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.19.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "bc68b257041ef3cbc95abe946a54f69b" }, { "dataPath": "params_shard_38.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.2.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "96ee8e329c66e71303856240b850a854" }, { "dataPath": "params_shard_39.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.2.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "287cbd3672f6456495f9f0388cd00ebd" }, { "dataPath": "params_shard_40.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.19.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.2.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.2.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.2.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.2.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.2.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.2.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "e8f214a9c7b6b29cc31d92fa41dea50d" }, { "dataPath": "params_shard_41.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.20.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "8e6645323550eef19f6d76c6058ce8e3" }, { "dataPath": "params_shard_42.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.20.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "8e018ed3531c005f58fe7e70146c5172" }, { "dataPath": "params_shard_43.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.2.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.20.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.20.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.20.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.20.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.20.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.20.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "46a472f2192f8f07db7174ac0445e81e" }, { "dataPath": "params_shard_44.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.21.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "9ffaf5430d17166ddddb46541e38a052" }, { "dataPath": "params_shard_45.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.21.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "9e40d634a0fc2ad35cf4030fbbdabc11" }, { "dataPath": "params_shard_46.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.20.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.21.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.21.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.21.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.21.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.21.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.21.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "16df28feb9c1f5e7a4d8e6f68862572b" }, { "dataPath": "params_shard_47.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.22.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "56f467ad10ab61c433da8587e95dc725" }, { "dataPath": "params_shard_48.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.22.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "0e6a1dac99710057507bd0d11b50eab4" }, { "dataPath": "params_shard_49.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.21.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.22.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.22.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.22.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.22.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.22.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.22.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "c0d30bda2866f2ea1f5c9d2edb6c3d31" }, { "dataPath": "params_shard_50.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.23.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "8991cc0cc00127f6b02c64a354f488b0" }, { "dataPath": "params_shard_51.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.23.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "7d908cc3592d6ac0f579d9af9c65880b" }, { "dataPath": "params_shard_52.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.22.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.23.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.23.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.23.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.23.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.23.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.23.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "3791dde6c32459d61ce491d5c68fbe75" }, { "dataPath": "params_shard_53.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.3.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "eaef3c26179adfc8b0e0e6da6ef384dc" }, { "dataPath": "params_shard_54.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.3.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "53ce84bd9cefa35a76b68651e91bde4e" }, { "dataPath": "params_shard_55.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.23.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.3.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.3.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.3.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.3.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.3.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.3.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "9fd086d4255532fe0553aa78febfa3c5" }, { "dataPath": "params_shard_56.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.4.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "e5418b1796c0d0bf6110b60f658db040" }, { "dataPath": "params_shard_57.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.4.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "de06fb6e7027821c255741e483f720c3" }, { "dataPath": "params_shard_58.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.3.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.4.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.4.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.4.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.4.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.4.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.4.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "3500a1c3f75d576966cc0ce12060a58d" }, { "dataPath": "params_shard_59.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.5.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "86566a953401d3bc7dbdb52821737977" }, { "dataPath": "params_shard_60.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.5.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "4a8e7de14545f19604080f3675bd4e25" }, { "dataPath": "params_shard_61.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.4.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.5.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.5.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.5.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.5.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.5.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.5.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "a5e78848bfe56b5f399599ba13cff465" }, { "dataPath": "params_shard_62.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.6.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "4fc2f473360bc7e4074f1eaf56407815" }, { "dataPath": "params_shard_63.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.6.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "dd64a40d442250a3e3e8d5ab80969163" }, { "dataPath": "params_shard_64.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.5.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.6.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.6.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.6.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.6.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.6.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.6.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "6664340a7fa63a488e56d3922dc4e6fc" }, { "dataPath": "params_shard_65.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.7.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "0272f4828c8856671f67f0f8c801c9ea" }, { "dataPath": "params_shard_66.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.7.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "19d4ca5052117c93fa60acc136cdf3c4" }, { "dataPath": "params_shard_67.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.6.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.7.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.7.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.7.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.7.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.7.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.7.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "5f01cda40185eca631e4e4580e5c7d43" }, { "dataPath": "params_shard_68.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.8.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "13c46a4e7b4595ea9dbe84fd3231411f" }, { "dataPath": "params_shard_69.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.8.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "a0e98bc964fe5c3169ffed182e2c2f32" }, { "dataPath": "params_shard_70.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.7.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.8.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.8.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.8.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.8.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.8.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.8.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "e6e2237c4be82896f7d3e51e80d3e532" }, { "dataPath": "params_shard_71.bin", "format": "raw-shard", "nbytes": 46137344, "records": [ { "name": "model.layers.9.mlp.gate_up_proj.weight", "shape": [ 11264, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 46137344, "byteOffset": 0 } ], "md5sum": "bedc9b33733acfb4eb0e63c6ea0caf49" }, { "dataPath": "params_shard_72.bin", "format": "raw-shard", "nbytes": 25165824, "records": [ { "name": "model.layers.9.self_attn.qkv_proj.weight", "shape": [ 6144, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 25165824, "byteOffset": 0 } ], "md5sum": "e2a0d2d0e1cc5aabd2c6d3e4d2aee4f7" }, { "dataPath": "params_shard_73.bin", "format": "raw-shard", "nbytes": 31485952, "records": [ { "name": "model.layers.8.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.layers.9.input_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.layers.9.input_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 }, { "name": "model.layers.9.mlp.down_proj.weight", "shape": [ 2048, 5632 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 23068672, "byteOffset": 8396800 }, { "name": "model.layers.9.post_attention_layernorm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31465472 }, { "name": "model.layers.9.post_attention_layernorm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 31469568 }, { "name": "model.layers.9.self_attn.qkv_proj.bias", "shape": [ 6144 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 12288, "byteOffset": 31473664 } ], "md5sum": "67a9e7491bfc338ff5dd57b5edf571e6" }, { "dataPath": "params_shard_74.bin", "format": "raw-shard", "nbytes": 8396800, "records": [ { "name": "model.layers.9.self_attn.o_proj.weight", "shape": [ 2048, 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 8388608, "byteOffset": 0 }, { "name": "model.norm.bias", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8388608 }, { "name": "model.norm.weight", "shape": [ 2048 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 4096, "byteOffset": 8392704 } ], "md5sum": "85fa8f8180a156fa5d21e32c7413cce0" } ] }