stablelm-2-zephyr-1_6b-q0f32-MLC / ndarray-cache.json
Tlopex's picture
Initial commit
2b5d618 verified
{
"metadata": {
"ParamSize": 220,
"ParamBytes": 6578061312.0,
"BitsPerParam": 32.0
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 411041792,
"records": [
{
"name": "lm_head.weight",
"shape": [
100352,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 411041792,
"byteOffset": 0
}
],
"md5sum": "8feaa03c43296af6d56d6ae34d79ff49"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 411041792,
"records": [
{
"name": "model.embed_tokens.weight",
"shape": [
100352,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 411041792,
"byteOffset": 0
}
],
"md5sum": "6e70bbcc4f9a5ebec925c6ce74a85d6e"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.0.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "1ee2c21c4c42bbfd1a96723eac48ed6f"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.0.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "8650fd58ae045e16af2dc1f9fdbc7c60"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 23068672,
"records": [
{
"name": "model.layers.1.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 0
}
],
"md5sum": "10234dc2b68ee4e0828d1e7a8751562c"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.1.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "fa1fc18500fca5973ad17d3129d76d1c"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.1.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "b0aa899c673432f2ab23655f1779b287"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 31514624,
"records": [
{
"name": "model.layers.0.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 0
},
{
"name": "model.layers.0.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 4096
},
{
"name": "model.layers.0.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8192
},
{
"name": "model.layers.0.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 23076864
},
{
"name": "model.layers.0.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 23080960
},
{
"name": "model.layers.0.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 23085056
},
{
"name": "model.layers.0.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 23097344
},
{
"name": "model.layers.1.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31485952
},
{
"name": "model.layers.1.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31490048
},
{
"name": "model.layers.1.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31494144
},
{
"name": "model.layers.1.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31498240
},
{
"name": "model.layers.1.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31502336
}
],
"md5sum": "a366140336e0fc5625f067a9715bbf3f"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.10.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "76709d8055c55eff2616cd2ee3fbb222"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.10.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "2157fce09a8f8aaaa3762fcb9646afcf"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.1.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.10.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.10.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.10.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.10.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.10.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.10.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "f5bb6f1e81f74526d8d921862d150950"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.11.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "6366654f8345be2accfc828ff5be0a2f"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.11.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "ddeaf590f32dbad9675ffb4f161aa468"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.10.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.11.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.11.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.11.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.11.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.11.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.11.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "861caedc6a61e8de4eb1c73344b8b352"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.12.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "54f1342270651067c9698809b8130756"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.12.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "8ba339fa062866f94263c2f7ca7f7a6a"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.11.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.12.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.12.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.12.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.12.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.12.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.12.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "79585b00793455bac8b7f41eb03a6531"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.13.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "cb530c4b04e5523a06192b121beee591"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.13.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "42561efa3873c8352ea1400b11c4eb87"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.12.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.13.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.13.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.13.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.13.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.13.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.13.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "f270a9de3e40c9bb3289c47d3975e489"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.14.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "96ac57d821a2148d40bf5da25776db0a"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.14.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "535b088c9a11993bc5716e8202abba1e"
},
{
"dataPath": "params_shard_22.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.13.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.14.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.14.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.14.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.14.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.14.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.14.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "39bbf585ed946719daa1b63ca36ff72a"
},
{
"dataPath": "params_shard_23.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.15.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "cc11aba9e48fe3e0ff913800e04bbc3d"
},
{
"dataPath": "params_shard_24.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.15.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "b8aba12c02a13c086f49ba2c0d08f534"
},
{
"dataPath": "params_shard_25.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.14.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.15.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.15.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.15.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.15.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.15.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.15.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "c348dfe6bca6a8f5dd9f16a62cf76c4f"
},
{
"dataPath": "params_shard_26.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.16.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "627233672b35ee886132e8a36824eaa0"
},
{
"dataPath": "params_shard_27.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.16.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "87eac07c30d3da72dde134591c228061"
},
{
"dataPath": "params_shard_28.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.15.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.16.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.16.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.16.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.16.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.16.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.16.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "a30882c3d8928cc34082b5578e15e158"
},
{
"dataPath": "params_shard_29.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.17.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "29e98bbacba4fb105f50eb5fb6a587cf"
},
{
"dataPath": "params_shard_30.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.17.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "d62b14f6d9e4a43032ca3f11957aa408"
},
{
"dataPath": "params_shard_31.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.16.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.17.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.17.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.17.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.17.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.17.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.17.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "9042ac3354d0c7eadb8c49892880a7e2"
},
{
"dataPath": "params_shard_32.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.18.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "3f0701e33136482f785a5fa0393f91e5"
},
{
"dataPath": "params_shard_33.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.18.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "2edf3df335285fe75a275d71b0cf0e46"
},
{
"dataPath": "params_shard_34.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.17.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.18.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.18.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.18.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.18.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.18.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.18.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "6c50305e7eff773699910ef8c2469366"
},
{
"dataPath": "params_shard_35.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.19.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "4e28c8d694655bd6560df661988f48ea"
},
{
"dataPath": "params_shard_36.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.19.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "9bef8b9a825ce9864f1a65dff4be96ee"
},
{
"dataPath": "params_shard_37.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.18.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.19.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.19.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.19.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.19.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.19.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.19.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "bc68b257041ef3cbc95abe946a54f69b"
},
{
"dataPath": "params_shard_38.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.2.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "96ee8e329c66e71303856240b850a854"
},
{
"dataPath": "params_shard_39.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.2.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "287cbd3672f6456495f9f0388cd00ebd"
},
{
"dataPath": "params_shard_40.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.19.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.2.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.2.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.2.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.2.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.2.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.2.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "e8f214a9c7b6b29cc31d92fa41dea50d"
},
{
"dataPath": "params_shard_41.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.20.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "8e6645323550eef19f6d76c6058ce8e3"
},
{
"dataPath": "params_shard_42.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.20.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "8e018ed3531c005f58fe7e70146c5172"
},
{
"dataPath": "params_shard_43.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.2.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.20.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.20.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.20.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.20.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.20.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.20.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "46a472f2192f8f07db7174ac0445e81e"
},
{
"dataPath": "params_shard_44.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.21.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "9ffaf5430d17166ddddb46541e38a052"
},
{
"dataPath": "params_shard_45.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.21.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "9e40d634a0fc2ad35cf4030fbbdabc11"
},
{
"dataPath": "params_shard_46.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.20.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.21.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.21.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.21.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.21.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.21.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.21.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "16df28feb9c1f5e7a4d8e6f68862572b"
},
{
"dataPath": "params_shard_47.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.22.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "56f467ad10ab61c433da8587e95dc725"
},
{
"dataPath": "params_shard_48.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.22.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "0e6a1dac99710057507bd0d11b50eab4"
},
{
"dataPath": "params_shard_49.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.21.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.22.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.22.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.22.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.22.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.22.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.22.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "c0d30bda2866f2ea1f5c9d2edb6c3d31"
},
{
"dataPath": "params_shard_50.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.23.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "8991cc0cc00127f6b02c64a354f488b0"
},
{
"dataPath": "params_shard_51.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.23.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "7d908cc3592d6ac0f579d9af9c65880b"
},
{
"dataPath": "params_shard_52.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.22.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.23.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.23.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.23.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.23.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.23.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.23.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "3791dde6c32459d61ce491d5c68fbe75"
},
{
"dataPath": "params_shard_53.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.3.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "eaef3c26179adfc8b0e0e6da6ef384dc"
},
{
"dataPath": "params_shard_54.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.3.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "53ce84bd9cefa35a76b68651e91bde4e"
},
{
"dataPath": "params_shard_55.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.23.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.3.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.3.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.3.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.3.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.3.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.3.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "9fd086d4255532fe0553aa78febfa3c5"
},
{
"dataPath": "params_shard_56.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.4.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "e5418b1796c0d0bf6110b60f658db040"
},
{
"dataPath": "params_shard_57.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.4.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "de06fb6e7027821c255741e483f720c3"
},
{
"dataPath": "params_shard_58.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.3.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.4.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.4.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.4.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.4.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.4.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.4.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "3500a1c3f75d576966cc0ce12060a58d"
},
{
"dataPath": "params_shard_59.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "86566a953401d3bc7dbdb52821737977"
},
{
"dataPath": "params_shard_60.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.5.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "4a8e7de14545f19604080f3675bd4e25"
},
{
"dataPath": "params_shard_61.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.4.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.5.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.5.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.5.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.5.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.5.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.5.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "a5e78848bfe56b5f399599ba13cff465"
},
{
"dataPath": "params_shard_62.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.6.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "4fc2f473360bc7e4074f1eaf56407815"
},
{
"dataPath": "params_shard_63.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.6.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "dd64a40d442250a3e3e8d5ab80969163"
},
{
"dataPath": "params_shard_64.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.5.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.6.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.6.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.6.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.6.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.6.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.6.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "6664340a7fa63a488e56d3922dc4e6fc"
},
{
"dataPath": "params_shard_65.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.7.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "0272f4828c8856671f67f0f8c801c9ea"
},
{
"dataPath": "params_shard_66.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.7.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "19d4ca5052117c93fa60acc136cdf3c4"
},
{
"dataPath": "params_shard_67.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.6.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.7.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.7.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.7.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.7.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.7.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.7.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "5f01cda40185eca631e4e4580e5c7d43"
},
{
"dataPath": "params_shard_68.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.8.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "13c46a4e7b4595ea9dbe84fd3231411f"
},
{
"dataPath": "params_shard_69.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.8.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "a0e98bc964fe5c3169ffed182e2c2f32"
},
{
"dataPath": "params_shard_70.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.7.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.8.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.8.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.8.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.8.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.8.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.8.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "e6e2237c4be82896f7d3e51e80d3e532"
},
{
"dataPath": "params_shard_71.bin",
"format": "raw-shard",
"nbytes": 46137344,
"records": [
{
"name": "model.layers.9.mlp.gate_up_proj.weight",
"shape": [
11264,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 46137344,
"byteOffset": 0
}
],
"md5sum": "bedc9b33733acfb4eb0e63c6ea0caf49"
},
{
"dataPath": "params_shard_72.bin",
"format": "raw-shard",
"nbytes": 25165824,
"records": [
{
"name": "model.layers.9.self_attn.qkv_proj.weight",
"shape": [
6144,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 25165824,
"byteOffset": 0
}
],
"md5sum": "e2a0d2d0e1cc5aabd2c6d3e4d2aee4f7"
},
{
"dataPath": "params_shard_73.bin",
"format": "raw-shard",
"nbytes": 31485952,
"records": [
{
"name": "model.layers.8.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.9.input_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.layers.9.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
},
{
"name": "model.layers.9.mlp.down_proj.weight",
"shape": [
2048,
5632
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 23068672,
"byteOffset": 8396800
},
{
"name": "model.layers.9.post_attention_layernorm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31465472
},
{
"name": "model.layers.9.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 31469568
},
{
"name": "model.layers.9.self_attn.qkv_proj.bias",
"shape": [
6144
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 12288,
"byteOffset": 31473664
}
],
"md5sum": "67a9e7491bfc338ff5dd57b5edf571e6"
},
{
"dataPath": "params_shard_74.bin",
"format": "raw-shard",
"nbytes": 8396800,
"records": [
{
"name": "model.layers.9.self_attn.o_proj.weight",
"shape": [
2048,
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.norm.bias",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8388608
},
{
"name": "model.norm.weight",
"shape": [
2048
],
"dtype": "float32",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 8392704
}
],
"md5sum": "85fa8f8180a156fa5d21e32c7413cce0"
}
]
}