Upload folder using huggingface_hub (#1)
Browse files- d0047398d55206e9d073722cc0644b3a20b001b4af1953a93bf91d3c1fa548de (75108c816cebaebf26d8c0012b51c1df573405c2)
- 15949045ea437f9d59ae3df133faad69acee6cc8b7a207d6ef9d12d07e1717b0 (251dffbf7d0b5cb12701fa24b4cb7fb91a0c81fb)
- 5ed3385dab19bd25abe1447102e1c921b53f38a184a682761ff33d84ca9cfcfa (c625eb96800ed97a6be25820049e4b4e9f64d633)
- 18d28d5619b00cea5c0bfbfc50cc29f50557b9416b6b1c8bc7bbeff300264a77 (9f4e277fb3be1b4d4f2f146c08ed530489f32efc)
- 52909b183fe575b757f315ddd53bc4c360f500be59e2e3776a4f160f353f47ee (6d79255fba7b0236e8c1cbc44d73907939766503)
- 9a6275f4f3446e2ca78ee008d4469d8f10eacbfe4aa7e6a663b17093f1fcf888 (34410e8db305e25278d87b76513147f747259203)
- d00e3e76cf2bc9b8b02cba67b9f13f0daca6f93db5f6b28494297d2a18dbf46c (0d786db128514fdca58ec14ee5670c4b47d27ab3)
- 904ae874b5bd601368740ef26b73bf57d125f0d3f5f63a82f195d70833dac87a (68bd4e75765f8bc0f1f0eeabe0fbe3995f533eb8)
- 429eeb5c0f2906b388d02b485881450f68e4b96cd5e99cc1bff46569f0eaf76a (85d67c6fdcd0875240dd0200fd930365f609d461)
- 125851b35e532570f511cf27ced0637918eef8a6884e277c1a1fa1e2208c9fe4 (55358a7664cd2bfad74dcee4a1a8635fc5f323d5)
- 0bc6fc8329fb67e986763009a0b52f56056d403dd6f0914fb9a86793bf225dca (7a33b905509e73925ae510f38d081ca72b8ac411)
- 04fb4127cd143b23f012718811700aaf0e6555c82bb6f57e20d0f3d603985147 (66e0ba7550965db68e3a7686a30db5206d62aad1)
- 0ad3800ab45a1d7ea3e8a21b6a06c002a00a3fd50791958e05126d66e40e81fa (64d39a6a56a2b3265b75d972b4a81aaf6769bb05)
- 47ccc4d67cdafbac2b1f68324151c1bc8512984b4ae914a51ae7e131c8e1ae5f (321a8ef0640d20639198afd949c8e33033f68f7c)
- 08b8cf65afad4bc67e86dcf9edcbf7e9d625bb101eac708a40efcd3fb2d74baa (084ac4fa89e251f1ee22b16619c6ff16ec93c5e4)
- 55214bed01c32d622bddd04947dee6eb639662e3ef7c1aa3db95eededea82b99 (e3161370b0c8bc9c692c6ee700b49be5bf8147d6)
- 86af7d3accfb0775b7a7eb3f82faf23cf014fcc46c474341bf61e9acc5c69f6d (f72d3cef690472394d5ebe0a5acff8f652f74324)
- cb0256bb37dc5c339b38f6caeef23ef156551e97fbb4077ae6e9b6c4d117b144 (7ad9b453762a289e3f7422a8c96b5a3cfe098427)
- df2c4324282068bff147379591a35fb94382be276f078b9819104bed49c3a937 (73a5307169c52110877a163684462e3a84bab7a7)
- dc37ff78dbcedadbc8560b9a06a52f5388137a14ff2cc6f89d7c246aae0e565f (3f70e92456787dedea5a0c70671872df4c453e0a)
- 5e14ee732df5f7890ff2191f1cd5df46e385e3550469c497dba6826f2d5714c1 (e93a3b933edd2ddbbb8b213bd9bbc0d009829bf4)
- 3c44b29fcd60081d20d177248bd1f4926f289bbdf7821fef039135b63a9b58ae (ff28f7de7de19a9371a5f9b8b7284a81ef3cadc2)
- ec99b4b99ffe7054072a2b6a3f70b9b42886e07448e4527c582e59cbcac954be (9995bdc93bee0a73c221386eab753ca5f4485e76)
- a698a38b471b8185ede9b77e4f5d59ccbd81bf8dd87162d0f60bab643b5dec89 (7e9c21727a71132159c1b07820564ab503859ec5)
- 9625211491852b37e82682fdd29c8b3f7d5781866a3ed786cf1e297a0234ac90 (a0e748d80694e2e45de14ca7e1bb30b7df7d4fb2)
- a22d1cc64d1557187e52c8e78d1c94da76e94e31f393c4644e32c758257c34d5 (e21817842d1c48cd6cc440e72c5991c41a1c34b6)
- 352d98ca94660918897be8321ca6460c8f61c30402ac1f145516967b78a95313 (abf87dfcae010511921c82395c41c45c0d8224b6)
- 0c78ff3b5993b68ad5ed8a51432e909297bba74cc2ee37ee04858119faf52dab (ecbf71bc63444b8e5755076e656ce03034049062)
- 24f681d52a01bf4607cacc89bfca65aa037bf375df1b39d06f503b497c9450f2 (3d9f0d8926a0c5ba63fecf6743c0cb798540593f)
- cdd06d5d28d732b2d20f892a0b452ef21bf823bef9efb4da228f82a0df65e76c (acb45bc8cdf7f6d1272aaa4262f6a06df7906348)
- ae26a273544c91fc01a1ad441d2a02dc7844d3a2f32d33b2adc311c2fdb10f86 (354ba26aa8c173d82254056b0cdf53043a69e175)
- 9beb9aa47eeb7ea8814f04e36f9bc6c058cb999d3a29171a545d21c04671525f (a635ae3fc26eafa78459fa77d9f76d01122762bf)
- 0c52d98ae22ce98dd9db61642fc2489b4227e06db61eb89a541381f5053d40df (c53321e6b8892a7224b02ed9d8a114dc80595900)
- 7f5575b1609aa6a3e3e408e2096cbbced0a9ca3835445443f159b3f13335085b (93b9e8cd7e65022b063a972e2e394bf91bec57f9)
- 37db0d9a0e4ffa256770404e69a785655bf7a8c9bb24e8101d2ae6d802e33b81 (ca8eaaf6b0827b07faf87d3f9f84b0d21c2d7840)
- config.json +27 -0
- pytorch_model-01-of-35.bin +3 -0
- pytorch_model-02-of-35.bin +3 -0
- pytorch_model-03-of-35.bin +3 -0
- pytorch_model-04-of-35.bin +3 -0
- pytorch_model-05-of-35.bin +3 -0
- pytorch_model-06-of-35.bin +3 -0
- pytorch_model-07-of-35.bin +3 -0
- pytorch_model-08-of-35.bin +3 -0
- pytorch_model-09-of-35.bin +3 -0
- pytorch_model-10-of-35.bin +3 -0
- pytorch_model-11-of-35.bin +3 -0
- pytorch_model-12-of-35.bin +3 -0
- pytorch_model-13-of-35.bin +3 -0
- pytorch_model-14-of-35.bin +3 -0
- pytorch_model-15-of-35.bin +3 -0
- pytorch_model-16-of-35.bin +3 -0
- pytorch_model-17-of-35.bin +3 -0
- pytorch_model-18-of-35.bin +3 -0
- pytorch_model-19-of-35.bin +3 -0
- pytorch_model-20-of-35.bin +3 -0
- pytorch_model-21-of-35.bin +3 -0
- pytorch_model-22-of-35.bin +3 -0
- pytorch_model-23-of-35.bin +3 -0
- pytorch_model-24-of-35.bin +3 -0
- pytorch_model-25-of-35.bin +3 -0
- pytorch_model-26-of-35.bin +3 -0
- pytorch_model-27-of-35.bin +3 -0
- pytorch_model-28-of-35.bin +3 -0
- pytorch_model-29-of-35.bin +3 -0
- pytorch_model-30-of-35.bin +3 -0
- pytorch_model-31-of-35.bin +3 -0
- pytorch_model-32-of-35.bin +3 -0
- pytorch_model-33-of-35.bin +3 -0
- pytorch_model-34-of-35.bin +3 -0
- pytorch_model-35-of-35.bin +3 -0
- pytorch_model.bin.index.json +1 -0
- pytorch_model.bin.sambatensor_index.json +1 -0
- special_tokens_map.json +23 -0
- tokenizer.json +0 -0
- tokenizer.model +3 -0
- tokenizer_config.json +33 -0
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"_name_or_path": "/import/ml-sc-nlpcheckpoints-scratch3/bol/llama70b_sql_finetune_decaylr_on_pretrain_4kgreedydrop_10epoch/step_668",
|
3 |
+
"architectures": [
|
4 |
+
"LlamaForCausalLM"
|
5 |
+
],
|
6 |
+
"bos_token_id": 1,
|
7 |
+
"eos_token_id": 2,
|
8 |
+
"hidden_act": "silu",
|
9 |
+
"hidden_size": 8192,
|
10 |
+
"initializer_range": 0.02,
|
11 |
+
"intermediate_size": 28672,
|
12 |
+
"max_position_embeddings": 4096,
|
13 |
+
"model_type": "llama",
|
14 |
+
"num_attention_heads": 64,
|
15 |
+
"num_hidden_layers": 80,
|
16 |
+
"num_key_value_heads": 8,
|
17 |
+
"pad_token_id": 0,
|
18 |
+
"pretraining_tp": 1,
|
19 |
+
"return_dict": false,
|
20 |
+
"rms_norm_eps": 1e-05,
|
21 |
+
"rope_scaling": null,
|
22 |
+
"tie_word_embeddings": false,
|
23 |
+
"torch_dtype": "float16",
|
24 |
+
"transformers_version": "4.31.0",
|
25 |
+
"use_cache": true,
|
26 |
+
"vocab_size": 32000
|
27 |
+
}
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:b5d276eab962ee2b13022d2544caad3cfe083fe71f530c77ba26c501ed82fe53
|
3 |
+
size 4248903383
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:5ba0b400fbb7a833ade9d394552e77e0ac10b2169db89322b6c5c253def67749
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:039a96d3eff13642d0f18f6fc8c3e4fc0679b349c3cb5df0549739efce2b3fdd
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:da1d522697adabcc2c2b0db87f0049565b16181e9c655ddb9394ae15db47e1ed
|
3 |
+
size 4194410769
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:dbb88f4dda77c7faf33cbccabbc499ced2cf11c5f59b96d536cb56d85efa8957
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:3686ec11ffdb1bebd41cc7dd921aff251630acc89c666804ae525f76f92581d7
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:7fe3344c403fb884edbbd7d9f856e06bce3a164ccd9e71a954d4080ca4b7ffad
|
3 |
+
size 4194410769
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:d7e1988c877a4b175cee347ca8e7b15a693119587d3ce983b8547f5aa665ab1e
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:d3dcd54d0783b2714d468c4e2e1772769c9d42555f11822dbdc0da79d90fe7ab
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:a467f51f48355d9cea9cbad734132f7899b26aa0f5b8b7ffe5f4af4590f41e82
|
3 |
+
size 4194410769
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:0b31688686809a1c379cfe3c915f450e7f8d12faf4ab3211a2fa7e76eb22b023
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:1dcb2b43b69fe805dd87352ac3c79f5951b7855897bb77cf77254d627a8a15e6
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:49604e711e51d7952565a33e0ed4b263f7ee0bfce1a8444048c4b7160ff4aabd
|
3 |
+
size 4194410769
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:40fae71562979506d0a55819b308a04327d39de369da9818ec2f9bdb159260c6
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:08528f0b5fc8bd5d6206b22ea1f72d70b95b39465a683ed2e6ca69e3502c83a5
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:cf530acc2f76afcb5d26d2315149f05e324fba9464868361ffc8eed240e76c8d
|
3 |
+
size 4194410769
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:4aaef6b77ac8c525103b832b8dcb94157d40ba4e6c6c043ad138f34ca5f947a4
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:753f502f490ee36e2f46dc2e2bba79533c4cca0cf552620ae5011bfe55666cfe
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:846d93428fd353393ddd1be1dc35d730a5194f01cc5614fa839b07bf61614cb8
|
3 |
+
size 4194410769
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:d5601d8893061fb1a1ec896a222b28ba1c4df8970a08923827f085fa724cf3f6
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:a1194194806efdd02c4161aa679c4365e62897f9d0e1242e5faa09d9f038927c
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:2b1d50a672bc29d156b2d1181ef547ce3c71f43af7c78351c2b17604cb14b85f
|
3 |
+
size 4194410769
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:788f962ebd6cc4e055f4a01697a32f94399d88d898826292889af53bdb0ee41e
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:797176b11a493bdf19d6d828b9989aa83a8f3638b1b5fd7014f187c91f2a1f8b
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:63cf10de4b331360b53ce710fd6379ea870347dc7206808b1f6c8ae6d70cfdda
|
3 |
+
size 4194410769
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:cf10e601780334049893b8aa9f281b15ed0a64bf2a2eba935237d3d144d7ca6b
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:913cf1692b925e981f66579b0f2addaab90c18ba25cdd2175d629f3ac84f5cde
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:137508f34f6cca7993a39bf55f1dbc217e53b15ed6e1a019e0847ee241ee877f
|
3 |
+
size 4194410769
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:028b1702a3c1abd293e388d705351a2648d7cb857bf700c4ee54042c449e525e
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:13d7bb8032dad55d47012134a7e21d36adbca483228c5d7cc704b8975bedcd17
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:d1c16a2825e025522f696ff63b0f1683fc870cafdcee13c9e0ed4649b9bd75ae
|
3 |
+
size 4194410769
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:9ee970651c7c1c308fc58e809c83ac4d941e9aee0b7c7fac57f36f8082c227f9
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:0cd49811efeb64bbe7d1912112d4843eacc82e4acad72ceeede9c0bccc7d7bde
|
3 |
+
size 3892385958
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:e4f896b03c73cee854bacb286065a7488302495aa13fa24cbfff81595fb589cd
|
3 |
+
size 4194410769
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:0b1606230a1b0e28a3b89b4e790ade456affe9af3ca3a3e8bcf78d3f9382e0a1
|
3 |
+
size 1933625427
|
@@ -0,0 +1 @@
|
|
|
|
|
1 |
+
{"metadata": {}, "weight_map": {"model.embed_tokens.weight": "pytorch_model-01-of-35.bin", "model.layers.0.self_attn.q_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.0.self_attn.k_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.0.self_attn.v_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.0.self_attn.o_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.0.self_attn.rotary_emb.inv_freq": "pytorch_model-01-of-35.bin", "model.layers.0.mlp.gate_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.0.mlp.up_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.0.mlp.down_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.0.input_layernorm.weight": "pytorch_model-01-of-35.bin", "model.layers.0.post_attention_layernorm.weight": "pytorch_model-01-of-35.bin", "model.layers.1.self_attn.q_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.1.self_attn.k_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.1.self_attn.v_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.1.self_attn.o_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.1.self_attn.rotary_emb.inv_freq": "pytorch_model-01-of-35.bin", "model.layers.1.mlp.gate_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.1.mlp.up_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.1.mlp.down_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.1.input_layernorm.weight": "pytorch_model-01-of-35.bin", "model.layers.1.post_attention_layernorm.weight": "pytorch_model-01-of-35.bin", "model.layers.2.self_attn.q_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.2.self_attn.k_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.2.self_attn.v_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.2.self_attn.o_proj.weight": "pytorch_model-01-of-35.bin", "model.layers.2.self_attn.rotary_emb.inv_freq": "pytorch_model-01-of-35.bin", "model.layers.2.mlp.gate_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.2.mlp.up_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.2.mlp.down_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.2.input_layernorm.weight": "pytorch_model-02-of-35.bin", "model.layers.2.post_attention_layernorm.weight": "pytorch_model-02-of-35.bin", "model.layers.3.self_attn.q_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.3.self_attn.k_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.3.self_attn.v_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.3.self_attn.o_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.3.self_attn.rotary_emb.inv_freq": "pytorch_model-02-of-35.bin", "model.layers.3.mlp.gate_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.3.mlp.up_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.3.mlp.down_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.3.input_layernorm.weight": "pytorch_model-02-of-35.bin", "model.layers.3.post_attention_layernorm.weight": "pytorch_model-02-of-35.bin", "model.layers.4.self_attn.q_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.4.self_attn.k_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.4.self_attn.v_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.4.self_attn.o_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.4.self_attn.rotary_emb.inv_freq": "pytorch_model-02-of-35.bin", "model.layers.4.mlp.gate_proj.weight": "pytorch_model-02-of-35.bin", "model.layers.4.mlp.up_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.4.mlp.down_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.4.input_layernorm.weight": "pytorch_model-03-of-35.bin", "model.layers.4.post_attention_layernorm.weight": "pytorch_model-03-of-35.bin", "model.layers.5.self_attn.q_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.5.self_attn.k_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.5.self_attn.v_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.5.self_attn.o_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.5.self_attn.rotary_emb.inv_freq": "pytorch_model-03-of-35.bin", "model.layers.5.mlp.gate_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.5.mlp.up_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.5.mlp.down_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.5.input_layernorm.weight": "pytorch_model-03-of-35.bin", "model.layers.5.post_attention_layernorm.weight": "pytorch_model-03-of-35.bin", "model.layers.6.self_attn.q_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.6.self_attn.k_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.6.self_attn.v_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.6.self_attn.o_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.6.self_attn.rotary_emb.inv_freq": "pytorch_model-03-of-35.bin", "model.layers.6.mlp.gate_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.6.mlp.up_proj.weight": "pytorch_model-03-of-35.bin", "model.layers.6.mlp.down_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.6.input_layernorm.weight": "pytorch_model-04-of-35.bin", "model.layers.6.post_attention_layernorm.weight": "pytorch_model-04-of-35.bin", "model.layers.7.self_attn.q_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.7.self_attn.k_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.7.self_attn.v_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.7.self_attn.o_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.7.self_attn.rotary_emb.inv_freq": "pytorch_model-04-of-35.bin", "model.layers.7.mlp.gate_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.7.mlp.up_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.7.mlp.down_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.7.input_layernorm.weight": "pytorch_model-04-of-35.bin", "model.layers.7.post_attention_layernorm.weight": "pytorch_model-04-of-35.bin", "model.layers.8.self_attn.q_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.8.self_attn.k_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.8.self_attn.v_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.8.self_attn.o_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.8.self_attn.rotary_emb.inv_freq": "pytorch_model-04-of-35.bin", "model.layers.8.mlp.gate_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.8.mlp.up_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.8.mlp.down_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.8.input_layernorm.weight": "pytorch_model-04-of-35.bin", "model.layers.8.post_attention_layernorm.weight": "pytorch_model-04-of-35.bin", "model.layers.9.self_attn.q_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.9.self_attn.k_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.9.self_attn.v_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.9.self_attn.o_proj.weight": "pytorch_model-04-of-35.bin", "model.layers.9.self_attn.rotary_emb.inv_freq": "pytorch_model-04-of-35.bin", "model.layers.9.mlp.gate_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.9.mlp.up_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.9.mlp.down_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.9.input_layernorm.weight": "pytorch_model-05-of-35.bin", "model.layers.9.post_attention_layernorm.weight": "pytorch_model-05-of-35.bin", "model.layers.10.self_attn.q_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.10.self_attn.k_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.10.self_attn.v_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.10.self_attn.o_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.10.self_attn.rotary_emb.inv_freq": "pytorch_model-05-of-35.bin", "model.layers.10.mlp.gate_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.10.mlp.up_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.10.mlp.down_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.10.input_layernorm.weight": "pytorch_model-05-of-35.bin", "model.layers.10.post_attention_layernorm.weight": "pytorch_model-05-of-35.bin", "model.layers.11.self_attn.q_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.11.self_attn.k_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.11.self_attn.v_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.11.self_attn.o_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.11.self_attn.rotary_emb.inv_freq": "pytorch_model-05-of-35.bin", "model.layers.11.mlp.gate_proj.weight": "pytorch_model-05-of-35.bin", "model.layers.11.mlp.up_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.11.mlp.down_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.11.input_layernorm.weight": "pytorch_model-06-of-35.bin", "model.layers.11.post_attention_layernorm.weight": "pytorch_model-06-of-35.bin", "model.layers.12.self_attn.q_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.12.self_attn.k_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.12.self_attn.v_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.12.self_attn.o_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.12.self_attn.rotary_emb.inv_freq": "pytorch_model-06-of-35.bin", "model.layers.12.mlp.gate_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.12.mlp.up_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.12.mlp.down_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.12.input_layernorm.weight": "pytorch_model-06-of-35.bin", "model.layers.12.post_attention_layernorm.weight": "pytorch_model-06-of-35.bin", "model.layers.13.self_attn.q_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.13.self_attn.k_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.13.self_attn.v_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.13.self_attn.o_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.13.self_attn.rotary_emb.inv_freq": "pytorch_model-06-of-35.bin", "model.layers.13.mlp.gate_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.13.mlp.up_proj.weight": "pytorch_model-06-of-35.bin", "model.layers.13.mlp.down_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.13.input_layernorm.weight": "pytorch_model-07-of-35.bin", "model.layers.13.post_attention_layernorm.weight": "pytorch_model-07-of-35.bin", "model.layers.14.self_attn.q_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.14.self_attn.k_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.14.self_attn.v_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.14.self_attn.o_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.14.self_attn.rotary_emb.inv_freq": "pytorch_model-07-of-35.bin", "model.layers.14.mlp.gate_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.14.mlp.up_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.14.mlp.down_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.14.input_layernorm.weight": "pytorch_model-07-of-35.bin", "model.layers.14.post_attention_layernorm.weight": "pytorch_model-07-of-35.bin", "model.layers.15.self_attn.q_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.15.self_attn.k_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.15.self_attn.v_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.15.self_attn.o_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.15.self_attn.rotary_emb.inv_freq": "pytorch_model-07-of-35.bin", "model.layers.15.mlp.gate_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.15.mlp.up_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.15.mlp.down_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.15.input_layernorm.weight": "pytorch_model-07-of-35.bin", "model.layers.15.post_attention_layernorm.weight": "pytorch_model-07-of-35.bin", "model.layers.16.self_attn.q_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.16.self_attn.k_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.16.self_attn.v_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.16.self_attn.o_proj.weight": "pytorch_model-07-of-35.bin", "model.layers.16.self_attn.rotary_emb.inv_freq": "pytorch_model-07-of-35.bin", "model.layers.16.mlp.gate_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.16.mlp.up_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.16.mlp.down_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.16.input_layernorm.weight": "pytorch_model-08-of-35.bin", "model.layers.16.post_attention_layernorm.weight": "pytorch_model-08-of-35.bin", "model.layers.17.self_attn.q_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.17.self_attn.k_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.17.self_attn.v_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.17.self_attn.o_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.17.self_attn.rotary_emb.inv_freq": "pytorch_model-08-of-35.bin", "model.layers.17.mlp.gate_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.17.mlp.up_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.17.mlp.down_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.17.input_layernorm.weight": "pytorch_model-08-of-35.bin", "model.layers.17.post_attention_layernorm.weight": "pytorch_model-08-of-35.bin", "model.layers.18.self_attn.q_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.18.self_attn.k_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.18.self_attn.v_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.18.self_attn.o_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.18.self_attn.rotary_emb.inv_freq": "pytorch_model-08-of-35.bin", "model.layers.18.mlp.gate_proj.weight": "pytorch_model-08-of-35.bin", "model.layers.18.mlp.up_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.18.mlp.down_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.18.input_layernorm.weight": "pytorch_model-09-of-35.bin", "model.layers.18.post_attention_layernorm.weight": "pytorch_model-09-of-35.bin", "model.layers.19.self_attn.q_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.19.self_attn.k_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.19.self_attn.v_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.19.self_attn.o_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.19.self_attn.rotary_emb.inv_freq": "pytorch_model-09-of-35.bin", "model.layers.19.mlp.gate_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.19.mlp.up_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.19.mlp.down_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.19.input_layernorm.weight": "pytorch_model-09-of-35.bin", "model.layers.19.post_attention_layernorm.weight": "pytorch_model-09-of-35.bin", "model.layers.20.self_attn.q_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.20.self_attn.k_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.20.self_attn.v_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.20.self_attn.o_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.20.self_attn.rotary_emb.inv_freq": "pytorch_model-09-of-35.bin", "model.layers.20.mlp.gate_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.20.mlp.up_proj.weight": "pytorch_model-09-of-35.bin", "model.layers.20.mlp.down_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.20.input_layernorm.weight": "pytorch_model-10-of-35.bin", "model.layers.20.post_attention_layernorm.weight": "pytorch_model-10-of-35.bin", "model.layers.21.self_attn.q_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.21.self_attn.k_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.21.self_attn.v_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.21.self_attn.o_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.21.self_attn.rotary_emb.inv_freq": "pytorch_model-10-of-35.bin", "model.layers.21.mlp.gate_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.21.mlp.up_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.21.mlp.down_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.21.input_layernorm.weight": "pytorch_model-10-of-35.bin", "model.layers.21.post_attention_layernorm.weight": "pytorch_model-10-of-35.bin", "model.layers.22.self_attn.q_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.22.self_attn.k_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.22.self_attn.v_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.22.self_attn.o_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.22.self_attn.rotary_emb.inv_freq": "pytorch_model-10-of-35.bin", "model.layers.22.mlp.gate_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.22.mlp.up_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.22.mlp.down_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.22.input_layernorm.weight": "pytorch_model-10-of-35.bin", "model.layers.22.post_attention_layernorm.weight": "pytorch_model-10-of-35.bin", "model.layers.23.self_attn.q_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.23.self_attn.k_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.23.self_attn.v_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.23.self_attn.o_proj.weight": "pytorch_model-10-of-35.bin", "model.layers.23.self_attn.rotary_emb.inv_freq": "pytorch_model-10-of-35.bin", "model.layers.23.mlp.gate_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.23.mlp.up_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.23.mlp.down_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.23.input_layernorm.weight": "pytorch_model-11-of-35.bin", "model.layers.23.post_attention_layernorm.weight": "pytorch_model-11-of-35.bin", "model.layers.24.self_attn.q_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.24.self_attn.k_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.24.self_attn.v_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.24.self_attn.o_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.24.self_attn.rotary_emb.inv_freq": "pytorch_model-11-of-35.bin", "model.layers.24.mlp.gate_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.24.mlp.up_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.24.mlp.down_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.24.input_layernorm.weight": "pytorch_model-11-of-35.bin", "model.layers.24.post_attention_layernorm.weight": "pytorch_model-11-of-35.bin", "model.layers.25.self_attn.q_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.25.self_attn.k_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.25.self_attn.v_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.25.self_attn.o_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.25.self_attn.rotary_emb.inv_freq": "pytorch_model-11-of-35.bin", "model.layers.25.mlp.gate_proj.weight": "pytorch_model-11-of-35.bin", "model.layers.25.mlp.up_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.25.mlp.down_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.25.input_layernorm.weight": "pytorch_model-12-of-35.bin", "model.layers.25.post_attention_layernorm.weight": "pytorch_model-12-of-35.bin", "model.layers.26.self_attn.q_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.26.self_attn.k_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.26.self_attn.v_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.26.self_attn.o_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.26.self_attn.rotary_emb.inv_freq": "pytorch_model-12-of-35.bin", "model.layers.26.mlp.gate_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.26.mlp.up_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.26.mlp.down_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.26.input_layernorm.weight": "pytorch_model-12-of-35.bin", "model.layers.26.post_attention_layernorm.weight": "pytorch_model-12-of-35.bin", "model.layers.27.self_attn.q_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.27.self_attn.k_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.27.self_attn.v_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.27.self_attn.o_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.27.self_attn.rotary_emb.inv_freq": "pytorch_model-12-of-35.bin", "model.layers.27.mlp.gate_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.27.mlp.up_proj.weight": "pytorch_model-12-of-35.bin", "model.layers.27.mlp.down_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.27.input_layernorm.weight": "pytorch_model-13-of-35.bin", "model.layers.27.post_attention_layernorm.weight": "pytorch_model-13-of-35.bin", "model.layers.28.self_attn.q_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.28.self_attn.k_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.28.self_attn.v_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.28.self_attn.o_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.28.self_attn.rotary_emb.inv_freq": "pytorch_model-13-of-35.bin", "model.layers.28.mlp.gate_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.28.mlp.up_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.28.mlp.down_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.28.input_layernorm.weight": "pytorch_model-13-of-35.bin", "model.layers.28.post_attention_layernorm.weight": "pytorch_model-13-of-35.bin", "model.layers.29.self_attn.q_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.29.self_attn.k_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.29.self_attn.v_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.29.self_attn.o_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.29.self_attn.rotary_emb.inv_freq": "pytorch_model-13-of-35.bin", "model.layers.29.mlp.gate_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.29.mlp.up_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.29.mlp.down_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.29.input_layernorm.weight": "pytorch_model-13-of-35.bin", "model.layers.29.post_attention_layernorm.weight": "pytorch_model-13-of-35.bin", "model.layers.30.self_attn.q_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.30.self_attn.k_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.30.self_attn.v_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.30.self_attn.o_proj.weight": "pytorch_model-13-of-35.bin", "model.layers.30.self_attn.rotary_emb.inv_freq": "pytorch_model-13-of-35.bin", "model.layers.30.mlp.gate_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.30.mlp.up_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.30.mlp.down_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.30.input_layernorm.weight": "pytorch_model-14-of-35.bin", "model.layers.30.post_attention_layernorm.weight": "pytorch_model-14-of-35.bin", "model.layers.31.self_attn.q_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.31.self_attn.k_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.31.self_attn.v_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.31.self_attn.o_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.31.self_attn.rotary_emb.inv_freq": "pytorch_model-14-of-35.bin", "model.layers.31.mlp.gate_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.31.mlp.up_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.31.mlp.down_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.31.input_layernorm.weight": "pytorch_model-14-of-35.bin", "model.layers.31.post_attention_layernorm.weight": "pytorch_model-14-of-35.bin", "model.layers.32.self_attn.q_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.32.self_attn.k_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.32.self_attn.v_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.32.self_attn.o_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.32.self_attn.rotary_emb.inv_freq": "pytorch_model-14-of-35.bin", "model.layers.32.mlp.gate_proj.weight": "pytorch_model-14-of-35.bin", "model.layers.32.mlp.up_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.32.mlp.down_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.32.input_layernorm.weight": "pytorch_model-15-of-35.bin", "model.layers.32.post_attention_layernorm.weight": "pytorch_model-15-of-35.bin", "model.layers.33.self_attn.q_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.33.self_attn.k_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.33.self_attn.v_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.33.self_attn.o_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.33.self_attn.rotary_emb.inv_freq": "pytorch_model-15-of-35.bin", "model.layers.33.mlp.gate_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.33.mlp.up_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.33.mlp.down_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.33.input_layernorm.weight": "pytorch_model-15-of-35.bin", "model.layers.33.post_attention_layernorm.weight": "pytorch_model-15-of-35.bin", "model.layers.34.self_attn.q_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.34.self_attn.k_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.34.self_attn.v_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.34.self_attn.o_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.34.self_attn.rotary_emb.inv_freq": "pytorch_model-15-of-35.bin", "model.layers.34.mlp.gate_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.34.mlp.up_proj.weight": "pytorch_model-15-of-35.bin", "model.layers.34.mlp.down_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.34.input_layernorm.weight": "pytorch_model-16-of-35.bin", "model.layers.34.post_attention_layernorm.weight": "pytorch_model-16-of-35.bin", "model.layers.35.self_attn.q_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.35.self_attn.k_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.35.self_attn.v_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.35.self_attn.o_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.35.self_attn.rotary_emb.inv_freq": "pytorch_model-16-of-35.bin", "model.layers.35.mlp.gate_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.35.mlp.up_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.35.mlp.down_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.35.input_layernorm.weight": "pytorch_model-16-of-35.bin", "model.layers.35.post_attention_layernorm.weight": "pytorch_model-16-of-35.bin", "model.layers.36.self_attn.q_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.36.self_attn.k_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.36.self_attn.v_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.36.self_attn.o_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.36.self_attn.rotary_emb.inv_freq": "pytorch_model-16-of-35.bin", "model.layers.36.mlp.gate_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.36.mlp.up_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.36.mlp.down_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.36.input_layernorm.weight": "pytorch_model-16-of-35.bin", "model.layers.36.post_attention_layernorm.weight": "pytorch_model-16-of-35.bin", "model.layers.37.self_attn.q_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.37.self_attn.k_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.37.self_attn.v_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.37.self_attn.o_proj.weight": "pytorch_model-16-of-35.bin", "model.layers.37.self_attn.rotary_emb.inv_freq": "pytorch_model-16-of-35.bin", "model.layers.37.mlp.gate_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.37.mlp.up_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.37.mlp.down_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.37.input_layernorm.weight": "pytorch_model-17-of-35.bin", "model.layers.37.post_attention_layernorm.weight": "pytorch_model-17-of-35.bin", "model.layers.38.self_attn.q_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.38.self_attn.k_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.38.self_attn.v_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.38.self_attn.o_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.38.self_attn.rotary_emb.inv_freq": "pytorch_model-17-of-35.bin", "model.layers.38.mlp.gate_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.38.mlp.up_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.38.mlp.down_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.38.input_layernorm.weight": "pytorch_model-17-of-35.bin", "model.layers.38.post_attention_layernorm.weight": "pytorch_model-17-of-35.bin", "model.layers.39.self_attn.q_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.39.self_attn.k_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.39.self_attn.v_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.39.self_attn.o_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.39.self_attn.rotary_emb.inv_freq": "pytorch_model-17-of-35.bin", "model.layers.39.mlp.gate_proj.weight": "pytorch_model-17-of-35.bin", "model.layers.39.mlp.up_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.39.mlp.down_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.39.input_layernorm.weight": "pytorch_model-18-of-35.bin", "model.layers.39.post_attention_layernorm.weight": "pytorch_model-18-of-35.bin", "model.layers.40.self_attn.q_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.40.self_attn.k_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.40.self_attn.v_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.40.self_attn.o_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.40.self_attn.rotary_emb.inv_freq": "pytorch_model-18-of-35.bin", "model.layers.40.mlp.gate_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.40.mlp.up_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.40.mlp.down_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.40.input_layernorm.weight": "pytorch_model-18-of-35.bin", "model.layers.40.post_attention_layernorm.weight": "pytorch_model-18-of-35.bin", "model.layers.41.self_attn.q_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.41.self_attn.k_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.41.self_attn.v_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.41.self_attn.o_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.41.self_attn.rotary_emb.inv_freq": "pytorch_model-18-of-35.bin", "model.layers.41.mlp.gate_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.41.mlp.up_proj.weight": "pytorch_model-18-of-35.bin", "model.layers.41.mlp.down_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.41.input_layernorm.weight": "pytorch_model-19-of-35.bin", "model.layers.41.post_attention_layernorm.weight": "pytorch_model-19-of-35.bin", "model.layers.42.self_attn.q_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.42.self_attn.k_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.42.self_attn.v_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.42.self_attn.o_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.42.self_attn.rotary_emb.inv_freq": "pytorch_model-19-of-35.bin", "model.layers.42.mlp.gate_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.42.mlp.up_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.42.mlp.down_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.42.input_layernorm.weight": "pytorch_model-19-of-35.bin", "model.layers.42.post_attention_layernorm.weight": "pytorch_model-19-of-35.bin", "model.layers.43.self_attn.q_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.43.self_attn.k_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.43.self_attn.v_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.43.self_attn.o_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.43.self_attn.rotary_emb.inv_freq": "pytorch_model-19-of-35.bin", "model.layers.43.mlp.gate_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.43.mlp.up_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.43.mlp.down_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.43.input_layernorm.weight": "pytorch_model-19-of-35.bin", "model.layers.43.post_attention_layernorm.weight": "pytorch_model-19-of-35.bin", "model.layers.44.self_attn.q_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.44.self_attn.k_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.44.self_attn.v_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.44.self_attn.o_proj.weight": "pytorch_model-19-of-35.bin", "model.layers.44.self_attn.rotary_emb.inv_freq": "pytorch_model-19-of-35.bin", "model.layers.44.mlp.gate_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.44.mlp.up_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.44.mlp.down_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.44.input_layernorm.weight": "pytorch_model-20-of-35.bin", "model.layers.44.post_attention_layernorm.weight": "pytorch_model-20-of-35.bin", "model.layers.45.self_attn.q_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.45.self_attn.k_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.45.self_attn.v_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.45.self_attn.o_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.45.self_attn.rotary_emb.inv_freq": "pytorch_model-20-of-35.bin", "model.layers.45.mlp.gate_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.45.mlp.up_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.45.mlp.down_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.45.input_layernorm.weight": "pytorch_model-20-of-35.bin", "model.layers.45.post_attention_layernorm.weight": "pytorch_model-20-of-35.bin", "model.layers.46.self_attn.q_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.46.self_attn.k_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.46.self_attn.v_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.46.self_attn.o_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.46.self_attn.rotary_emb.inv_freq": "pytorch_model-20-of-35.bin", "model.layers.46.mlp.gate_proj.weight": "pytorch_model-20-of-35.bin", "model.layers.46.mlp.up_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.46.mlp.down_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.46.input_layernorm.weight": "pytorch_model-21-of-35.bin", "model.layers.46.post_attention_layernorm.weight": "pytorch_model-21-of-35.bin", "model.layers.47.self_attn.q_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.47.self_attn.k_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.47.self_attn.v_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.47.self_attn.o_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.47.self_attn.rotary_emb.inv_freq": "pytorch_model-21-of-35.bin", "model.layers.47.mlp.gate_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.47.mlp.up_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.47.mlp.down_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.47.input_layernorm.weight": "pytorch_model-21-of-35.bin", "model.layers.47.post_attention_layernorm.weight": "pytorch_model-21-of-35.bin", "model.layers.48.self_attn.q_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.48.self_attn.k_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.48.self_attn.v_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.48.self_attn.o_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.48.self_attn.rotary_emb.inv_freq": "pytorch_model-21-of-35.bin", "model.layers.48.mlp.gate_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.48.mlp.up_proj.weight": "pytorch_model-21-of-35.bin", "model.layers.48.mlp.down_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.48.input_layernorm.weight": "pytorch_model-22-of-35.bin", "model.layers.48.post_attention_layernorm.weight": "pytorch_model-22-of-35.bin", "model.layers.49.self_attn.q_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.49.self_attn.k_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.49.self_attn.v_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.49.self_attn.o_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.49.self_attn.rotary_emb.inv_freq": "pytorch_model-22-of-35.bin", "model.layers.49.mlp.gate_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.49.mlp.up_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.49.mlp.down_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.49.input_layernorm.weight": "pytorch_model-22-of-35.bin", "model.layers.49.post_attention_layernorm.weight": "pytorch_model-22-of-35.bin", "model.layers.50.self_attn.q_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.50.self_attn.k_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.50.self_attn.v_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.50.self_attn.o_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.50.self_attn.rotary_emb.inv_freq": "pytorch_model-22-of-35.bin", "model.layers.50.mlp.gate_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.50.mlp.up_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.50.mlp.down_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.50.input_layernorm.weight": "pytorch_model-22-of-35.bin", "model.layers.50.post_attention_layernorm.weight": "pytorch_model-22-of-35.bin", "model.layers.51.self_attn.q_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.51.self_attn.k_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.51.self_attn.v_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.51.self_attn.o_proj.weight": "pytorch_model-22-of-35.bin", "model.layers.51.self_attn.rotary_emb.inv_freq": "pytorch_model-22-of-35.bin", "model.layers.51.mlp.gate_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.51.mlp.up_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.51.mlp.down_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.51.input_layernorm.weight": "pytorch_model-23-of-35.bin", "model.layers.51.post_attention_layernorm.weight": "pytorch_model-23-of-35.bin", "model.layers.52.self_attn.q_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.52.self_attn.k_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.52.self_attn.v_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.52.self_attn.o_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.52.self_attn.rotary_emb.inv_freq": "pytorch_model-23-of-35.bin", "model.layers.52.mlp.gate_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.52.mlp.up_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.52.mlp.down_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.52.input_layernorm.weight": "pytorch_model-23-of-35.bin", "model.layers.52.post_attention_layernorm.weight": "pytorch_model-23-of-35.bin", "model.layers.53.self_attn.q_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.53.self_attn.k_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.53.self_attn.v_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.53.self_attn.o_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.53.self_attn.rotary_emb.inv_freq": "pytorch_model-23-of-35.bin", "model.layers.53.mlp.gate_proj.weight": "pytorch_model-23-of-35.bin", "model.layers.53.mlp.up_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.53.mlp.down_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.53.input_layernorm.weight": "pytorch_model-24-of-35.bin", "model.layers.53.post_attention_layernorm.weight": "pytorch_model-24-of-35.bin", "model.layers.54.self_attn.q_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.54.self_attn.k_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.54.self_attn.v_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.54.self_attn.o_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.54.self_attn.rotary_emb.inv_freq": "pytorch_model-24-of-35.bin", "model.layers.54.mlp.gate_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.54.mlp.up_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.54.mlp.down_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.54.input_layernorm.weight": "pytorch_model-24-of-35.bin", "model.layers.54.post_attention_layernorm.weight": "pytorch_model-24-of-35.bin", "model.layers.55.self_attn.q_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.55.self_attn.k_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.55.self_attn.v_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.55.self_attn.o_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.55.self_attn.rotary_emb.inv_freq": "pytorch_model-24-of-35.bin", "model.layers.55.mlp.gate_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.55.mlp.up_proj.weight": "pytorch_model-24-of-35.bin", "model.layers.55.mlp.down_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.55.input_layernorm.weight": "pytorch_model-25-of-35.bin", "model.layers.55.post_attention_layernorm.weight": "pytorch_model-25-of-35.bin", "model.layers.56.self_attn.q_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.56.self_attn.k_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.56.self_attn.v_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.56.self_attn.o_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.56.self_attn.rotary_emb.inv_freq": "pytorch_model-25-of-35.bin", "model.layers.56.mlp.gate_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.56.mlp.up_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.56.mlp.down_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.56.input_layernorm.weight": "pytorch_model-25-of-35.bin", "model.layers.56.post_attention_layernorm.weight": "pytorch_model-25-of-35.bin", "model.layers.57.self_attn.q_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.57.self_attn.k_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.57.self_attn.v_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.57.self_attn.o_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.57.self_attn.rotary_emb.inv_freq": "pytorch_model-25-of-35.bin", "model.layers.57.mlp.gate_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.57.mlp.up_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.57.mlp.down_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.57.input_layernorm.weight": "pytorch_model-25-of-35.bin", "model.layers.57.post_attention_layernorm.weight": "pytorch_model-25-of-35.bin", "model.layers.58.self_attn.q_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.58.self_attn.k_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.58.self_attn.v_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.58.self_attn.o_proj.weight": "pytorch_model-25-of-35.bin", "model.layers.58.self_attn.rotary_emb.inv_freq": "pytorch_model-25-of-35.bin", "model.layers.58.mlp.gate_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.58.mlp.up_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.58.mlp.down_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.58.input_layernorm.weight": "pytorch_model-26-of-35.bin", "model.layers.58.post_attention_layernorm.weight": "pytorch_model-26-of-35.bin", "model.layers.59.self_attn.q_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.59.self_attn.k_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.59.self_attn.v_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.59.self_attn.o_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.59.self_attn.rotary_emb.inv_freq": "pytorch_model-26-of-35.bin", "model.layers.59.mlp.gate_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.59.mlp.up_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.59.mlp.down_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.59.input_layernorm.weight": "pytorch_model-26-of-35.bin", "model.layers.59.post_attention_layernorm.weight": "pytorch_model-26-of-35.bin", "model.layers.60.self_attn.q_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.60.self_attn.k_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.60.self_attn.v_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.60.self_attn.o_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.60.self_attn.rotary_emb.inv_freq": "pytorch_model-26-of-35.bin", "model.layers.60.mlp.gate_proj.weight": "pytorch_model-26-of-35.bin", "model.layers.60.mlp.up_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.60.mlp.down_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.60.input_layernorm.weight": "pytorch_model-27-of-35.bin", "model.layers.60.post_attention_layernorm.weight": "pytorch_model-27-of-35.bin", "model.layers.61.self_attn.q_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.61.self_attn.k_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.61.self_attn.v_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.61.self_attn.o_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.61.self_attn.rotary_emb.inv_freq": "pytorch_model-27-of-35.bin", "model.layers.61.mlp.gate_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.61.mlp.up_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.61.mlp.down_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.61.input_layernorm.weight": "pytorch_model-27-of-35.bin", "model.layers.61.post_attention_layernorm.weight": "pytorch_model-27-of-35.bin", "model.layers.62.self_attn.q_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.62.self_attn.k_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.62.self_attn.v_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.62.self_attn.o_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.62.self_attn.rotary_emb.inv_freq": "pytorch_model-27-of-35.bin", "model.layers.62.mlp.gate_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.62.mlp.up_proj.weight": "pytorch_model-27-of-35.bin", "model.layers.62.mlp.down_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.62.input_layernorm.weight": "pytorch_model-28-of-35.bin", "model.layers.62.post_attention_layernorm.weight": "pytorch_model-28-of-35.bin", "model.layers.63.self_attn.q_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.63.self_attn.k_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.63.self_attn.v_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.63.self_attn.o_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.63.self_attn.rotary_emb.inv_freq": "pytorch_model-28-of-35.bin", "model.layers.63.mlp.gate_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.63.mlp.up_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.63.mlp.down_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.63.input_layernorm.weight": "pytorch_model-28-of-35.bin", "model.layers.63.post_attention_layernorm.weight": "pytorch_model-28-of-35.bin", "model.layers.64.self_attn.q_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.64.self_attn.k_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.64.self_attn.v_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.64.self_attn.o_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.64.self_attn.rotary_emb.inv_freq": "pytorch_model-28-of-35.bin", "model.layers.64.mlp.gate_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.64.mlp.up_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.64.mlp.down_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.64.input_layernorm.weight": "pytorch_model-28-of-35.bin", "model.layers.64.post_attention_layernorm.weight": "pytorch_model-28-of-35.bin", "model.layers.65.self_attn.q_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.65.self_attn.k_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.65.self_attn.v_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.65.self_attn.o_proj.weight": "pytorch_model-28-of-35.bin", "model.layers.65.self_attn.rotary_emb.inv_freq": "pytorch_model-28-of-35.bin", "model.layers.65.mlp.gate_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.65.mlp.up_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.65.mlp.down_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.65.input_layernorm.weight": "pytorch_model-29-of-35.bin", "model.layers.65.post_attention_layernorm.weight": "pytorch_model-29-of-35.bin", "model.layers.66.self_attn.q_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.66.self_attn.k_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.66.self_attn.v_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.66.self_attn.o_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.66.self_attn.rotary_emb.inv_freq": "pytorch_model-29-of-35.bin", "model.layers.66.mlp.gate_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.66.mlp.up_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.66.mlp.down_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.66.input_layernorm.weight": "pytorch_model-29-of-35.bin", "model.layers.66.post_attention_layernorm.weight": "pytorch_model-29-of-35.bin", "model.layers.67.self_attn.q_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.67.self_attn.k_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.67.self_attn.v_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.67.self_attn.o_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.67.self_attn.rotary_emb.inv_freq": "pytorch_model-29-of-35.bin", "model.layers.67.mlp.gate_proj.weight": "pytorch_model-29-of-35.bin", "model.layers.67.mlp.up_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.67.mlp.down_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.67.input_layernorm.weight": "pytorch_model-30-of-35.bin", "model.layers.67.post_attention_layernorm.weight": "pytorch_model-30-of-35.bin", "model.layers.68.self_attn.q_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.68.self_attn.k_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.68.self_attn.v_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.68.self_attn.o_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.68.self_attn.rotary_emb.inv_freq": "pytorch_model-30-of-35.bin", "model.layers.68.mlp.gate_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.68.mlp.up_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.68.mlp.down_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.68.input_layernorm.weight": "pytorch_model-30-of-35.bin", "model.layers.68.post_attention_layernorm.weight": "pytorch_model-30-of-35.bin", "model.layers.69.self_attn.q_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.69.self_attn.k_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.69.self_attn.v_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.69.self_attn.o_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.69.self_attn.rotary_emb.inv_freq": "pytorch_model-30-of-35.bin", "model.layers.69.mlp.gate_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.69.mlp.up_proj.weight": "pytorch_model-30-of-35.bin", "model.layers.69.mlp.down_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.69.input_layernorm.weight": "pytorch_model-31-of-35.bin", "model.layers.69.post_attention_layernorm.weight": "pytorch_model-31-of-35.bin", "model.layers.70.self_attn.q_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.70.self_attn.k_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.70.self_attn.v_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.70.self_attn.o_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.70.self_attn.rotary_emb.inv_freq": "pytorch_model-31-of-35.bin", "model.layers.70.mlp.gate_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.70.mlp.up_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.70.mlp.down_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.70.input_layernorm.weight": "pytorch_model-31-of-35.bin", "model.layers.70.post_attention_layernorm.weight": "pytorch_model-31-of-35.bin", "model.layers.71.self_attn.q_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.71.self_attn.k_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.71.self_attn.v_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.71.self_attn.o_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.71.self_attn.rotary_emb.inv_freq": "pytorch_model-31-of-35.bin", "model.layers.71.mlp.gate_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.71.mlp.up_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.71.mlp.down_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.71.input_layernorm.weight": "pytorch_model-31-of-35.bin", "model.layers.71.post_attention_layernorm.weight": "pytorch_model-31-of-35.bin", "model.layers.72.self_attn.q_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.72.self_attn.k_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.72.self_attn.v_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.72.self_attn.o_proj.weight": "pytorch_model-31-of-35.bin", "model.layers.72.self_attn.rotary_emb.inv_freq": "pytorch_model-31-of-35.bin", "model.layers.72.mlp.gate_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.72.mlp.up_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.72.mlp.down_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.72.input_layernorm.weight": "pytorch_model-32-of-35.bin", "model.layers.72.post_attention_layernorm.weight": "pytorch_model-32-of-35.bin", "model.layers.73.self_attn.q_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.73.self_attn.k_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.73.self_attn.v_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.73.self_attn.o_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.73.self_attn.rotary_emb.inv_freq": "pytorch_model-32-of-35.bin", "model.layers.73.mlp.gate_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.73.mlp.up_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.73.mlp.down_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.73.input_layernorm.weight": "pytorch_model-32-of-35.bin", "model.layers.73.post_attention_layernorm.weight": "pytorch_model-32-of-35.bin", "model.layers.74.self_attn.q_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.74.self_attn.k_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.74.self_attn.v_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.74.self_attn.o_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.74.self_attn.rotary_emb.inv_freq": "pytorch_model-32-of-35.bin", "model.layers.74.mlp.gate_proj.weight": "pytorch_model-32-of-35.bin", "model.layers.74.mlp.up_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.74.mlp.down_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.74.input_layernorm.weight": "pytorch_model-33-of-35.bin", "model.layers.74.post_attention_layernorm.weight": "pytorch_model-33-of-35.bin", "model.layers.75.self_attn.q_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.75.self_attn.k_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.75.self_attn.v_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.75.self_attn.o_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.75.self_attn.rotary_emb.inv_freq": "pytorch_model-33-of-35.bin", "model.layers.75.mlp.gate_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.75.mlp.up_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.75.mlp.down_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.75.input_layernorm.weight": "pytorch_model-33-of-35.bin", "model.layers.75.post_attention_layernorm.weight": "pytorch_model-33-of-35.bin", "model.layers.76.self_attn.q_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.76.self_attn.k_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.76.self_attn.v_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.76.self_attn.o_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.76.self_attn.rotary_emb.inv_freq": "pytorch_model-33-of-35.bin", "model.layers.76.mlp.gate_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.76.mlp.up_proj.weight": "pytorch_model-33-of-35.bin", "model.layers.76.mlp.down_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.76.input_layernorm.weight": "pytorch_model-34-of-35.bin", "model.layers.76.post_attention_layernorm.weight": "pytorch_model-34-of-35.bin", "model.layers.77.self_attn.q_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.77.self_attn.k_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.77.self_attn.v_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.77.self_attn.o_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.77.self_attn.rotary_emb.inv_freq": "pytorch_model-34-of-35.bin", "model.layers.77.mlp.gate_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.77.mlp.up_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.77.mlp.down_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.77.input_layernorm.weight": "pytorch_model-34-of-35.bin", "model.layers.77.post_attention_layernorm.weight": "pytorch_model-34-of-35.bin", "model.layers.78.self_attn.q_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.78.self_attn.k_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.78.self_attn.v_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.78.self_attn.o_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.78.self_attn.rotary_emb.inv_freq": "pytorch_model-34-of-35.bin", "model.layers.78.mlp.gate_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.78.mlp.up_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.78.mlp.down_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.78.input_layernorm.weight": "pytorch_model-34-of-35.bin", "model.layers.78.post_attention_layernorm.weight": "pytorch_model-34-of-35.bin", "model.layers.79.self_attn.q_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.79.self_attn.k_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.79.self_attn.v_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.79.self_attn.o_proj.weight": "pytorch_model-34-of-35.bin", "model.layers.79.self_attn.rotary_emb.inv_freq": "pytorch_model-34-of-35.bin", "model.layers.79.mlp.gate_proj.weight": "pytorch_model-35-of-35.bin", "model.layers.79.mlp.up_proj.weight": "pytorch_model-35-of-35.bin", "model.layers.79.mlp.down_proj.weight": "pytorch_model-35-of-35.bin", "model.layers.79.input_layernorm.weight": "pytorch_model-35-of-35.bin", "model.layers.79.post_attention_layernorm.weight": "pytorch_model-35-of-35.bin", "model.norm.weight": "pytorch_model-35-of-35.bin", "lm_head.weight": "pytorch_model-35-of-35.bin"}}
|
@@ -0,0 +1 @@
|
|
|
|
|
1 |
+
{"model.embed_tokens.weight": "llamaforcausallm__model__embed_tokens__weight", "model.layers.0.self_attn.q_proj.weight": "llamaforcausallm__model__layers__0__self_attn__q_proj__weight", "model.layers.0.self_attn.k_proj.weight": "llamaforcausallm__model__layers__0__self_attn__k_proj__weight", "model.layers.0.self_attn.v_proj.weight": "llamaforcausallm__model__layers__0__self_attn__v_proj__weight", "model.layers.0.self_attn.o_proj.weight": "llamaforcausallm__model__layers__0__self_attn__o_proj__weight", "model.layers.0.mlp.gate_proj.weight": "llamaforcausallm__model__layers__0__mlp__gate_proj__weight", "model.layers.0.mlp.up_proj.weight": "llamaforcausallm__model__layers__0__mlp__up_proj__weight", "model.layers.0.mlp.down_proj.weight": "llamaforcausallm__model__layers__0__mlp__down_proj__weight", "model.layers.0.input_layernorm.weight": "llamaforcausallm__model__layers__0__input_layernorm__weight", "model.layers.0.post_attention_layernorm.weight": "llamaforcausallm__model__layers__0__post_attention_layernorm__weight", "model.layers.1.self_attn.q_proj.weight": "llamaforcausallm__model__layers__1__self_attn__q_proj__weight", "model.layers.1.self_attn.k_proj.weight": "llamaforcausallm__model__layers__1__self_attn__k_proj__weight", "model.layers.1.self_attn.v_proj.weight": "llamaforcausallm__model__layers__1__self_attn__v_proj__weight", "model.layers.1.self_attn.o_proj.weight": "llamaforcausallm__model__layers__1__self_attn__o_proj__weight", "model.layers.1.mlp.gate_proj.weight": "llamaforcausallm__model__layers__1__mlp__gate_proj__weight", "model.layers.1.mlp.up_proj.weight": "llamaforcausallm__model__layers__1__mlp__up_proj__weight", "model.layers.1.mlp.down_proj.weight": "llamaforcausallm__model__layers__1__mlp__down_proj__weight", "model.layers.1.input_layernorm.weight": "llamaforcausallm__model__layers__1__input_layernorm__weight", "model.layers.1.post_attention_layernorm.weight": "llamaforcausallm__model__layers__1__post_attention_layernorm__weight", "model.layers.2.self_attn.q_proj.weight": "llamaforcausallm__model__layers__2__self_attn__q_proj__weight", "model.layers.2.self_attn.k_proj.weight": "llamaforcausallm__model__layers__2__self_attn__k_proj__weight", "model.layers.2.self_attn.v_proj.weight": "llamaforcausallm__model__layers__2__self_attn__v_proj__weight", "model.layers.2.self_attn.o_proj.weight": "llamaforcausallm__model__layers__2__self_attn__o_proj__weight", "model.layers.2.mlp.gate_proj.weight": "llamaforcausallm__model__layers__2__mlp__gate_proj__weight", "model.layers.2.mlp.up_proj.weight": "llamaforcausallm__model__layers__2__mlp__up_proj__weight", "model.layers.2.mlp.down_proj.weight": "llamaforcausallm__model__layers__2__mlp__down_proj__weight", "model.layers.2.input_layernorm.weight": "llamaforcausallm__model__layers__2__input_layernorm__weight", "model.layers.2.post_attention_layernorm.weight": "llamaforcausallm__model__layers__2__post_attention_layernorm__weight", "model.layers.3.self_attn.q_proj.weight": "llamaforcausallm__model__layers__3__self_attn__q_proj__weight", "model.layers.3.self_attn.k_proj.weight": "llamaforcausallm__model__layers__3__self_attn__k_proj__weight", "model.layers.3.self_attn.v_proj.weight": "llamaforcausallm__model__layers__3__self_attn__v_proj__weight", "model.layers.3.self_attn.o_proj.weight": "llamaforcausallm__model__layers__3__self_attn__o_proj__weight", "model.layers.3.mlp.gate_proj.weight": "llamaforcausallm__model__layers__3__mlp__gate_proj__weight", "model.layers.3.mlp.up_proj.weight": "llamaforcausallm__model__layers__3__mlp__up_proj__weight", "model.layers.3.mlp.down_proj.weight": "llamaforcausallm__model__layers__3__mlp__down_proj__weight", "model.layers.3.input_layernorm.weight": "llamaforcausallm__model__layers__3__input_layernorm__weight", "model.layers.3.post_attention_layernorm.weight": "llamaforcausallm__model__layers__3__post_attention_layernorm__weight", "model.layers.4.self_attn.q_proj.weight": "llamaforcausallm__model__layers__4__self_attn__q_proj__weight", "model.layers.4.self_attn.k_proj.weight": "llamaforcausallm__model__layers__4__self_attn__k_proj__weight", "model.layers.4.self_attn.v_proj.weight": "llamaforcausallm__model__layers__4__self_attn__v_proj__weight", "model.layers.4.self_attn.o_proj.weight": "llamaforcausallm__model__layers__4__self_attn__o_proj__weight", "model.layers.4.mlp.gate_proj.weight": "llamaforcausallm__model__layers__4__mlp__gate_proj__weight", "model.layers.4.mlp.up_proj.weight": "llamaforcausallm__model__layers__4__mlp__up_proj__weight", "model.layers.4.mlp.down_proj.weight": "llamaforcausallm__model__layers__4__mlp__down_proj__weight", "model.layers.4.input_layernorm.weight": "llamaforcausallm__model__layers__4__input_layernorm__weight", "model.layers.4.post_attention_layernorm.weight": "llamaforcausallm__model__layers__4__post_attention_layernorm__weight", "model.layers.5.self_attn.q_proj.weight": "llamaforcausallm__model__layers__5__self_attn__q_proj__weight", "model.layers.5.self_attn.k_proj.weight": "llamaforcausallm__model__layers__5__self_attn__k_proj__weight", "model.layers.5.self_attn.v_proj.weight": "llamaforcausallm__model__layers__5__self_attn__v_proj__weight", "model.layers.5.self_attn.o_proj.weight": "llamaforcausallm__model__layers__5__self_attn__o_proj__weight", "model.layers.5.mlp.gate_proj.weight": "llamaforcausallm__model__layers__5__mlp__gate_proj__weight", "model.layers.5.mlp.up_proj.weight": "llamaforcausallm__model__layers__5__mlp__up_proj__weight", "model.layers.5.mlp.down_proj.weight": "llamaforcausallm__model__layers__5__mlp__down_proj__weight", "model.layers.5.input_layernorm.weight": "llamaforcausallm__model__layers__5__input_layernorm__weight", "model.layers.5.post_attention_layernorm.weight": "llamaforcausallm__model__layers__5__post_attention_layernorm__weight", "model.layers.6.self_attn.q_proj.weight": "llamaforcausallm__model__layers__6__self_attn__q_proj__weight", "model.layers.6.self_attn.k_proj.weight": "llamaforcausallm__model__layers__6__self_attn__k_proj__weight", "model.layers.6.self_attn.v_proj.weight": "llamaforcausallm__model__layers__6__self_attn__v_proj__weight", "model.layers.6.self_attn.o_proj.weight": "llamaforcausallm__model__layers__6__self_attn__o_proj__weight", "model.layers.6.mlp.gate_proj.weight": "llamaforcausallm__model__layers__6__mlp__gate_proj__weight", "model.layers.6.mlp.up_proj.weight": "llamaforcausallm__model__layers__6__mlp__up_proj__weight", "model.layers.6.mlp.down_proj.weight": "llamaforcausallm__model__layers__6__mlp__down_proj__weight", "model.layers.6.input_layernorm.weight": "llamaforcausallm__model__layers__6__input_layernorm__weight", "model.layers.6.post_attention_layernorm.weight": "llamaforcausallm__model__layers__6__post_attention_layernorm__weight", "model.layers.7.self_attn.q_proj.weight": "llamaforcausallm__model__layers__7__self_attn__q_proj__weight", "model.layers.7.self_attn.k_proj.weight": "llamaforcausallm__model__layers__7__self_attn__k_proj__weight", "model.layers.7.self_attn.v_proj.weight": "llamaforcausallm__model__layers__7__self_attn__v_proj__weight", "model.layers.7.self_attn.o_proj.weight": "llamaforcausallm__model__layers__7__self_attn__o_proj__weight", "model.layers.7.mlp.gate_proj.weight": "llamaforcausallm__model__layers__7__mlp__gate_proj__weight", "model.layers.7.mlp.up_proj.weight": "llamaforcausallm__model__layers__7__mlp__up_proj__weight", "model.layers.7.mlp.down_proj.weight": "llamaforcausallm__model__layers__7__mlp__down_proj__weight", "model.layers.7.input_layernorm.weight": "llamaforcausallm__model__layers__7__input_layernorm__weight", "model.layers.7.post_attention_layernorm.weight": "llamaforcausallm__model__layers__7__post_attention_layernorm__weight", "model.layers.8.self_attn.q_proj.weight": "llamaforcausallm__model__layers__8__self_attn__q_proj__weight", "model.layers.8.self_attn.k_proj.weight": "llamaforcausallm__model__layers__8__self_attn__k_proj__weight", "model.layers.8.self_attn.v_proj.weight": "llamaforcausallm__model__layers__8__self_attn__v_proj__weight", "model.layers.8.self_attn.o_proj.weight": "llamaforcausallm__model__layers__8__self_attn__o_proj__weight", "model.layers.8.mlp.gate_proj.weight": "llamaforcausallm__model__layers__8__mlp__gate_proj__weight", "model.layers.8.mlp.up_proj.weight": "llamaforcausallm__model__layers__8__mlp__up_proj__weight", "model.layers.8.mlp.down_proj.weight": "llamaforcausallm__model__layers__8__mlp__down_proj__weight", "model.layers.8.input_layernorm.weight": "llamaforcausallm__model__layers__8__input_layernorm__weight", "model.layers.8.post_attention_layernorm.weight": "llamaforcausallm__model__layers__8__post_attention_layernorm__weight", "model.layers.9.self_attn.q_proj.weight": "llamaforcausallm__model__layers__9__self_attn__q_proj__weight", "model.layers.9.self_attn.k_proj.weight": "llamaforcausallm__model__layers__9__self_attn__k_proj__weight", "model.layers.9.self_attn.v_proj.weight": "llamaforcausallm__model__layers__9__self_attn__v_proj__weight", "model.layers.9.self_attn.o_proj.weight": "llamaforcausallm__model__layers__9__self_attn__o_proj__weight", "model.layers.9.mlp.gate_proj.weight": "llamaforcausallm__model__layers__9__mlp__gate_proj__weight", "model.layers.9.mlp.up_proj.weight": "llamaforcausallm__model__layers__9__mlp__up_proj__weight", "model.layers.9.mlp.down_proj.weight": "llamaforcausallm__model__layers__9__mlp__down_proj__weight", "model.layers.9.input_layernorm.weight": "llamaforcausallm__model__layers__9__input_layernorm__weight", "model.layers.9.post_attention_layernorm.weight": "llamaforcausallm__model__layers__9__post_attention_layernorm__weight", "model.layers.10.self_attn.q_proj.weight": "llamaforcausallm__model__layers__10__self_attn__q_proj__weight", "model.layers.10.self_attn.k_proj.weight": "llamaforcausallm__model__layers__10__self_attn__k_proj__weight", "model.layers.10.self_attn.v_proj.weight": "llamaforcausallm__model__layers__10__self_attn__v_proj__weight", "model.layers.10.self_attn.o_proj.weight": "llamaforcausallm__model__layers__10__self_attn__o_proj__weight", "model.layers.10.mlp.gate_proj.weight": "llamaforcausallm__model__layers__10__mlp__gate_proj__weight", "model.layers.10.mlp.up_proj.weight": "llamaforcausallm__model__layers__10__mlp__up_proj__weight", "model.layers.10.mlp.down_proj.weight": "llamaforcausallm__model__layers__10__mlp__down_proj__weight", "model.layers.10.input_layernorm.weight": "llamaforcausallm__model__layers__10__input_layernorm__weight", "model.layers.10.post_attention_layernorm.weight": "llamaforcausallm__model__layers__10__post_attention_layernorm__weight", "model.layers.11.self_attn.q_proj.weight": "llamaforcausallm__model__layers__11__self_attn__q_proj__weight", "model.layers.11.self_attn.k_proj.weight": "llamaforcausallm__model__layers__11__self_attn__k_proj__weight", "model.layers.11.self_attn.v_proj.weight": "llamaforcausallm__model__layers__11__self_attn__v_proj__weight", "model.layers.11.self_attn.o_proj.weight": "llamaforcausallm__model__layers__11__self_attn__o_proj__weight", "model.layers.11.mlp.gate_proj.weight": "llamaforcausallm__model__layers__11__mlp__gate_proj__weight", "model.layers.11.mlp.up_proj.weight": "llamaforcausallm__model__layers__11__mlp__up_proj__weight", "model.layers.11.mlp.down_proj.weight": "llamaforcausallm__model__layers__11__mlp__down_proj__weight", "model.layers.11.input_layernorm.weight": "llamaforcausallm__model__layers__11__input_layernorm__weight", "model.layers.11.post_attention_layernorm.weight": "llamaforcausallm__model__layers__11__post_attention_layernorm__weight", "model.layers.12.self_attn.q_proj.weight": "llamaforcausallm__model__layers__12__self_attn__q_proj__weight", "model.layers.12.self_attn.k_proj.weight": "llamaforcausallm__model__layers__12__self_attn__k_proj__weight", "model.layers.12.self_attn.v_proj.weight": "llamaforcausallm__model__layers__12__self_attn__v_proj__weight", "model.layers.12.self_attn.o_proj.weight": "llamaforcausallm__model__layers__12__self_attn__o_proj__weight", "model.layers.12.mlp.gate_proj.weight": "llamaforcausallm__model__layers__12__mlp__gate_proj__weight", "model.layers.12.mlp.up_proj.weight": "llamaforcausallm__model__layers__12__mlp__up_proj__weight", "model.layers.12.mlp.down_proj.weight": "llamaforcausallm__model__layers__12__mlp__down_proj__weight", "model.layers.12.input_layernorm.weight": "llamaforcausallm__model__layers__12__input_layernorm__weight", "model.layers.12.post_attention_layernorm.weight": "llamaforcausallm__model__layers__12__post_attention_layernorm__weight", "model.layers.13.self_attn.q_proj.weight": "llamaforcausallm__model__layers__13__self_attn__q_proj__weight", "model.layers.13.self_attn.k_proj.weight": "llamaforcausallm__model__layers__13__self_attn__k_proj__weight", "model.layers.13.self_attn.v_proj.weight": "llamaforcausallm__model__layers__13__self_attn__v_proj__weight", "model.layers.13.self_attn.o_proj.weight": "llamaforcausallm__model__layers__13__self_attn__o_proj__weight", "model.layers.13.mlp.gate_proj.weight": "llamaforcausallm__model__layers__13__mlp__gate_proj__weight", "model.layers.13.mlp.up_proj.weight": "llamaforcausallm__model__layers__13__mlp__up_proj__weight", "model.layers.13.mlp.down_proj.weight": "llamaforcausallm__model__layers__13__mlp__down_proj__weight", "model.layers.13.input_layernorm.weight": "llamaforcausallm__model__layers__13__input_layernorm__weight", "model.layers.13.post_attention_layernorm.weight": "llamaforcausallm__model__layers__13__post_attention_layernorm__weight", "model.layers.14.self_attn.q_proj.weight": "llamaforcausallm__model__layers__14__self_attn__q_proj__weight", "model.layers.14.self_attn.k_proj.weight": "llamaforcausallm__model__layers__14__self_attn__k_proj__weight", "model.layers.14.self_attn.v_proj.weight": "llamaforcausallm__model__layers__14__self_attn__v_proj__weight", "model.layers.14.self_attn.o_proj.weight": "llamaforcausallm__model__layers__14__self_attn__o_proj__weight", "model.layers.14.mlp.gate_proj.weight": "llamaforcausallm__model__layers__14__mlp__gate_proj__weight", "model.layers.14.mlp.up_proj.weight": "llamaforcausallm__model__layers__14__mlp__up_proj__weight", "model.layers.14.mlp.down_proj.weight": "llamaforcausallm__model__layers__14__mlp__down_proj__weight", "model.layers.14.input_layernorm.weight": "llamaforcausallm__model__layers__14__input_layernorm__weight", "model.layers.14.post_attention_layernorm.weight": "llamaforcausallm__model__layers__14__post_attention_layernorm__weight", "model.layers.15.self_attn.q_proj.weight": "llamaforcausallm__model__layers__15__self_attn__q_proj__weight", "model.layers.15.self_attn.k_proj.weight": "llamaforcausallm__model__layers__15__self_attn__k_proj__weight", "model.layers.15.self_attn.v_proj.weight": "llamaforcausallm__model__layers__15__self_attn__v_proj__weight", "model.layers.15.self_attn.o_proj.weight": "llamaforcausallm__model__layers__15__self_attn__o_proj__weight", "model.layers.15.mlp.gate_proj.weight": "llamaforcausallm__model__layers__15__mlp__gate_proj__weight", "model.layers.15.mlp.up_proj.weight": "llamaforcausallm__model__layers__15__mlp__up_proj__weight", "model.layers.15.mlp.down_proj.weight": "llamaforcausallm__model__layers__15__mlp__down_proj__weight", "model.layers.15.input_layernorm.weight": "llamaforcausallm__model__layers__15__input_layernorm__weight", "model.layers.15.post_attention_layernorm.weight": "llamaforcausallm__model__layers__15__post_attention_layernorm__weight", "model.layers.16.self_attn.q_proj.weight": "llamaforcausallm__model__layers__16__self_attn__q_proj__weight", "model.layers.16.self_attn.k_proj.weight": "llamaforcausallm__model__layers__16__self_attn__k_proj__weight", "model.layers.16.self_attn.v_proj.weight": "llamaforcausallm__model__layers__16__self_attn__v_proj__weight", "model.layers.16.self_attn.o_proj.weight": "llamaforcausallm__model__layers__16__self_attn__o_proj__weight", "model.layers.16.mlp.gate_proj.weight": "llamaforcausallm__model__layers__16__mlp__gate_proj__weight", "model.layers.16.mlp.up_proj.weight": "llamaforcausallm__model__layers__16__mlp__up_proj__weight", "model.layers.16.mlp.down_proj.weight": "llamaforcausallm__model__layers__16__mlp__down_proj__weight", "model.layers.16.input_layernorm.weight": "llamaforcausallm__model__layers__16__input_layernorm__weight", "model.layers.16.post_attention_layernorm.weight": "llamaforcausallm__model__layers__16__post_attention_layernorm__weight", "model.layers.17.self_attn.q_proj.weight": "llamaforcausallm__model__layers__17__self_attn__q_proj__weight", "model.layers.17.self_attn.k_proj.weight": "llamaforcausallm__model__layers__17__self_attn__k_proj__weight", "model.layers.17.self_attn.v_proj.weight": "llamaforcausallm__model__layers__17__self_attn__v_proj__weight", "model.layers.17.self_attn.o_proj.weight": "llamaforcausallm__model__layers__17__self_attn__o_proj__weight", "model.layers.17.mlp.gate_proj.weight": "llamaforcausallm__model__layers__17__mlp__gate_proj__weight", "model.layers.17.mlp.up_proj.weight": "llamaforcausallm__model__layers__17__mlp__up_proj__weight", "model.layers.17.mlp.down_proj.weight": "llamaforcausallm__model__layers__17__mlp__down_proj__weight", "model.layers.17.input_layernorm.weight": "llamaforcausallm__model__layers__17__input_layernorm__weight", "model.layers.17.post_attention_layernorm.weight": "llamaforcausallm__model__layers__17__post_attention_layernorm__weight", "model.layers.18.self_attn.q_proj.weight": "llamaforcausallm__model__layers__18__self_attn__q_proj__weight", "model.layers.18.self_attn.k_proj.weight": "llamaforcausallm__model__layers__18__self_attn__k_proj__weight", "model.layers.18.self_attn.v_proj.weight": "llamaforcausallm__model__layers__18__self_attn__v_proj__weight", "model.layers.18.self_attn.o_proj.weight": "llamaforcausallm__model__layers__18__self_attn__o_proj__weight", "model.layers.18.mlp.gate_proj.weight": "llamaforcausallm__model__layers__18__mlp__gate_proj__weight", "model.layers.18.mlp.up_proj.weight": "llamaforcausallm__model__layers__18__mlp__up_proj__weight", "model.layers.18.mlp.down_proj.weight": "llamaforcausallm__model__layers__18__mlp__down_proj__weight", "model.layers.18.input_layernorm.weight": "llamaforcausallm__model__layers__18__input_layernorm__weight", "model.layers.18.post_attention_layernorm.weight": "llamaforcausallm__model__layers__18__post_attention_layernorm__weight", "model.layers.19.self_attn.q_proj.weight": "llamaforcausallm__model__layers__19__self_attn__q_proj__weight", "model.layers.19.self_attn.k_proj.weight": "llamaforcausallm__model__layers__19__self_attn__k_proj__weight", "model.layers.19.self_attn.v_proj.weight": "llamaforcausallm__model__layers__19__self_attn__v_proj__weight", "model.layers.19.self_attn.o_proj.weight": "llamaforcausallm__model__layers__19__self_attn__o_proj__weight", "model.layers.19.mlp.gate_proj.weight": "llamaforcausallm__model__layers__19__mlp__gate_proj__weight", "model.layers.19.mlp.up_proj.weight": "llamaforcausallm__model__layers__19__mlp__up_proj__weight", "model.layers.19.mlp.down_proj.weight": "llamaforcausallm__model__layers__19__mlp__down_proj__weight", "model.layers.19.input_layernorm.weight": "llamaforcausallm__model__layers__19__input_layernorm__weight", "model.layers.19.post_attention_layernorm.weight": "llamaforcausallm__model__layers__19__post_attention_layernorm__weight", "model.layers.20.self_attn.q_proj.weight": "llamaforcausallm__model__layers__20__self_attn__q_proj__weight", "model.layers.20.self_attn.k_proj.weight": "llamaforcausallm__model__layers__20__self_attn__k_proj__weight", "model.layers.20.self_attn.v_proj.weight": "llamaforcausallm__model__layers__20__self_attn__v_proj__weight", "model.layers.20.self_attn.o_proj.weight": "llamaforcausallm__model__layers__20__self_attn__o_proj__weight", "model.layers.20.mlp.gate_proj.weight": "llamaforcausallm__model__layers__20__mlp__gate_proj__weight", "model.layers.20.mlp.up_proj.weight": "llamaforcausallm__model__layers__20__mlp__up_proj__weight", "model.layers.20.mlp.down_proj.weight": "llamaforcausallm__model__layers__20__mlp__down_proj__weight", "model.layers.20.input_layernorm.weight": "llamaforcausallm__model__layers__20__input_layernorm__weight", "model.layers.20.post_attention_layernorm.weight": "llamaforcausallm__model__layers__20__post_attention_layernorm__weight", "model.layers.21.self_attn.q_proj.weight": "llamaforcausallm__model__layers__21__self_attn__q_proj__weight", "model.layers.21.self_attn.k_proj.weight": "llamaforcausallm__model__layers__21__self_attn__k_proj__weight", "model.layers.21.self_attn.v_proj.weight": "llamaforcausallm__model__layers__21__self_attn__v_proj__weight", "model.layers.21.self_attn.o_proj.weight": "llamaforcausallm__model__layers__21__self_attn__o_proj__weight", "model.layers.21.mlp.gate_proj.weight": "llamaforcausallm__model__layers__21__mlp__gate_proj__weight", "model.layers.21.mlp.up_proj.weight": "llamaforcausallm__model__layers__21__mlp__up_proj__weight", "model.layers.21.mlp.down_proj.weight": "llamaforcausallm__model__layers__21__mlp__down_proj__weight", "model.layers.21.input_layernorm.weight": "llamaforcausallm__model__layers__21__input_layernorm__weight", "model.layers.21.post_attention_layernorm.weight": "llamaforcausallm__model__layers__21__post_attention_layernorm__weight", "model.layers.22.self_attn.q_proj.weight": "llamaforcausallm__model__layers__22__self_attn__q_proj__weight", "model.layers.22.self_attn.k_proj.weight": "llamaforcausallm__model__layers__22__self_attn__k_proj__weight", "model.layers.22.self_attn.v_proj.weight": "llamaforcausallm__model__layers__22__self_attn__v_proj__weight", "model.layers.22.self_attn.o_proj.weight": "llamaforcausallm__model__layers__22__self_attn__o_proj__weight", "model.layers.22.mlp.gate_proj.weight": "llamaforcausallm__model__layers__22__mlp__gate_proj__weight", "model.layers.22.mlp.up_proj.weight": "llamaforcausallm__model__layers__22__mlp__up_proj__weight", "model.layers.22.mlp.down_proj.weight": "llamaforcausallm__model__layers__22__mlp__down_proj__weight", "model.layers.22.input_layernorm.weight": "llamaforcausallm__model__layers__22__input_layernorm__weight", "model.layers.22.post_attention_layernorm.weight": "llamaforcausallm__model__layers__22__post_attention_layernorm__weight", "model.layers.23.self_attn.q_proj.weight": "llamaforcausallm__model__layers__23__self_attn__q_proj__weight", "model.layers.23.self_attn.k_proj.weight": "llamaforcausallm__model__layers__23__self_attn__k_proj__weight", "model.layers.23.self_attn.v_proj.weight": "llamaforcausallm__model__layers__23__self_attn__v_proj__weight", "model.layers.23.self_attn.o_proj.weight": "llamaforcausallm__model__layers__23__self_attn__o_proj__weight", "model.layers.23.mlp.gate_proj.weight": "llamaforcausallm__model__layers__23__mlp__gate_proj__weight", "model.layers.23.mlp.up_proj.weight": "llamaforcausallm__model__layers__23__mlp__up_proj__weight", "model.layers.23.mlp.down_proj.weight": "llamaforcausallm__model__layers__23__mlp__down_proj__weight", "model.layers.23.input_layernorm.weight": "llamaforcausallm__model__layers__23__input_layernorm__weight", "model.layers.23.post_attention_layernorm.weight": "llamaforcausallm__model__layers__23__post_attention_layernorm__weight", "model.layers.24.self_attn.q_proj.weight": "llamaforcausallm__model__layers__24__self_attn__q_proj__weight", "model.layers.24.self_attn.k_proj.weight": "llamaforcausallm__model__layers__24__self_attn__k_proj__weight", "model.layers.24.self_attn.v_proj.weight": "llamaforcausallm__model__layers__24__self_attn__v_proj__weight", "model.layers.24.self_attn.o_proj.weight": "llamaforcausallm__model__layers__24__self_attn__o_proj__weight", "model.layers.24.mlp.gate_proj.weight": "llamaforcausallm__model__layers__24__mlp__gate_proj__weight", "model.layers.24.mlp.up_proj.weight": "llamaforcausallm__model__layers__24__mlp__up_proj__weight", "model.layers.24.mlp.down_proj.weight": "llamaforcausallm__model__layers__24__mlp__down_proj__weight", "model.layers.24.input_layernorm.weight": "llamaforcausallm__model__layers__24__input_layernorm__weight", "model.layers.24.post_attention_layernorm.weight": "llamaforcausallm__model__layers__24__post_attention_layernorm__weight", "model.layers.25.self_attn.q_proj.weight": "llamaforcausallm__model__layers__25__self_attn__q_proj__weight", "model.layers.25.self_attn.k_proj.weight": "llamaforcausallm__model__layers__25__self_attn__k_proj__weight", "model.layers.25.self_attn.v_proj.weight": "llamaforcausallm__model__layers__25__self_attn__v_proj__weight", "model.layers.25.self_attn.o_proj.weight": "llamaforcausallm__model__layers__25__self_attn__o_proj__weight", "model.layers.25.mlp.gate_proj.weight": "llamaforcausallm__model__layers__25__mlp__gate_proj__weight", "model.layers.25.mlp.up_proj.weight": "llamaforcausallm__model__layers__25__mlp__up_proj__weight", "model.layers.25.mlp.down_proj.weight": "llamaforcausallm__model__layers__25__mlp__down_proj__weight", "model.layers.25.input_layernorm.weight": "llamaforcausallm__model__layers__25__input_layernorm__weight", "model.layers.25.post_attention_layernorm.weight": "llamaforcausallm__model__layers__25__post_attention_layernorm__weight", "model.layers.26.self_attn.q_proj.weight": "llamaforcausallm__model__layers__26__self_attn__q_proj__weight", "model.layers.26.self_attn.k_proj.weight": "llamaforcausallm__model__layers__26__self_attn__k_proj__weight", "model.layers.26.self_attn.v_proj.weight": "llamaforcausallm__model__layers__26__self_attn__v_proj__weight", "model.layers.26.self_attn.o_proj.weight": "llamaforcausallm__model__layers__26__self_attn__o_proj__weight", "model.layers.26.mlp.gate_proj.weight": "llamaforcausallm__model__layers__26__mlp__gate_proj__weight", "model.layers.26.mlp.up_proj.weight": "llamaforcausallm__model__layers__26__mlp__up_proj__weight", "model.layers.26.mlp.down_proj.weight": "llamaforcausallm__model__layers__26__mlp__down_proj__weight", "model.layers.26.input_layernorm.weight": "llamaforcausallm__model__layers__26__input_layernorm__weight", "model.layers.26.post_attention_layernorm.weight": "llamaforcausallm__model__layers__26__post_attention_layernorm__weight", "model.layers.27.self_attn.q_proj.weight": "llamaforcausallm__model__layers__27__self_attn__q_proj__weight", "model.layers.27.self_attn.k_proj.weight": "llamaforcausallm__model__layers__27__self_attn__k_proj__weight", "model.layers.27.self_attn.v_proj.weight": "llamaforcausallm__model__layers__27__self_attn__v_proj__weight", "model.layers.27.self_attn.o_proj.weight": "llamaforcausallm__model__layers__27__self_attn__o_proj__weight", "model.layers.27.mlp.gate_proj.weight": "llamaforcausallm__model__layers__27__mlp__gate_proj__weight", "model.layers.27.mlp.up_proj.weight": "llamaforcausallm__model__layers__27__mlp__up_proj__weight", "model.layers.27.mlp.down_proj.weight": "llamaforcausallm__model__layers__27__mlp__down_proj__weight", "model.layers.27.input_layernorm.weight": "llamaforcausallm__model__layers__27__input_layernorm__weight", "model.layers.27.post_attention_layernorm.weight": "llamaforcausallm__model__layers__27__post_attention_layernorm__weight", "model.layers.28.self_attn.q_proj.weight": "llamaforcausallm__model__layers__28__self_attn__q_proj__weight", "model.layers.28.self_attn.k_proj.weight": "llamaforcausallm__model__layers__28__self_attn__k_proj__weight", "model.layers.28.self_attn.v_proj.weight": "llamaforcausallm__model__layers__28__self_attn__v_proj__weight", "model.layers.28.self_attn.o_proj.weight": "llamaforcausallm__model__layers__28__self_attn__o_proj__weight", "model.layers.28.mlp.gate_proj.weight": "llamaforcausallm__model__layers__28__mlp__gate_proj__weight", "model.layers.28.mlp.up_proj.weight": "llamaforcausallm__model__layers__28__mlp__up_proj__weight", "model.layers.28.mlp.down_proj.weight": "llamaforcausallm__model__layers__28__mlp__down_proj__weight", "model.layers.28.input_layernorm.weight": "llamaforcausallm__model__layers__28__input_layernorm__weight", "model.layers.28.post_attention_layernorm.weight": "llamaforcausallm__model__layers__28__post_attention_layernorm__weight", "model.layers.29.self_attn.q_proj.weight": "llamaforcausallm__model__layers__29__self_attn__q_proj__weight", "model.layers.29.self_attn.k_proj.weight": "llamaforcausallm__model__layers__29__self_attn__k_proj__weight", "model.layers.29.self_attn.v_proj.weight": "llamaforcausallm__model__layers__29__self_attn__v_proj__weight", "model.layers.29.self_attn.o_proj.weight": "llamaforcausallm__model__layers__29__self_attn__o_proj__weight", "model.layers.29.mlp.gate_proj.weight": "llamaforcausallm__model__layers__29__mlp__gate_proj__weight", "model.layers.29.mlp.up_proj.weight": "llamaforcausallm__model__layers__29__mlp__up_proj__weight", "model.layers.29.mlp.down_proj.weight": "llamaforcausallm__model__layers__29__mlp__down_proj__weight", "model.layers.29.input_layernorm.weight": "llamaforcausallm__model__layers__29__input_layernorm__weight", "model.layers.29.post_attention_layernorm.weight": "llamaforcausallm__model__layers__29__post_attention_layernorm__weight", "model.layers.30.self_attn.q_proj.weight": "llamaforcausallm__model__layers__30__self_attn__q_proj__weight", "model.layers.30.self_attn.k_proj.weight": "llamaforcausallm__model__layers__30__self_attn__k_proj__weight", "model.layers.30.self_attn.v_proj.weight": "llamaforcausallm__model__layers__30__self_attn__v_proj__weight", "model.layers.30.self_attn.o_proj.weight": "llamaforcausallm__model__layers__30__self_attn__o_proj__weight", "model.layers.30.mlp.gate_proj.weight": "llamaforcausallm__model__layers__30__mlp__gate_proj__weight", "model.layers.30.mlp.up_proj.weight": "llamaforcausallm__model__layers__30__mlp__up_proj__weight", "model.layers.30.mlp.down_proj.weight": "llamaforcausallm__model__layers__30__mlp__down_proj__weight", "model.layers.30.input_layernorm.weight": "llamaforcausallm__model__layers__30__input_layernorm__weight", "model.layers.30.post_attention_layernorm.weight": "llamaforcausallm__model__layers__30__post_attention_layernorm__weight", "model.layers.31.self_attn.q_proj.weight": "llamaforcausallm__model__layers__31__self_attn__q_proj__weight", "model.layers.31.self_attn.k_proj.weight": "llamaforcausallm__model__layers__31__self_attn__k_proj__weight", "model.layers.31.self_attn.v_proj.weight": "llamaforcausallm__model__layers__31__self_attn__v_proj__weight", "model.layers.31.self_attn.o_proj.weight": "llamaforcausallm__model__layers__31__self_attn__o_proj__weight", "model.layers.31.mlp.gate_proj.weight": "llamaforcausallm__model__layers__31__mlp__gate_proj__weight", "model.layers.31.mlp.up_proj.weight": "llamaforcausallm__model__layers__31__mlp__up_proj__weight", "model.layers.31.mlp.down_proj.weight": "llamaforcausallm__model__layers__31__mlp__down_proj__weight", "model.layers.31.input_layernorm.weight": "llamaforcausallm__model__layers__31__input_layernorm__weight", "model.layers.31.post_attention_layernorm.weight": "llamaforcausallm__model__layers__31__post_attention_layernorm__weight", "model.layers.32.self_attn.q_proj.weight": "llamaforcausallm__model__layers__32__self_attn__q_proj__weight", "model.layers.32.self_attn.k_proj.weight": "llamaforcausallm__model__layers__32__self_attn__k_proj__weight", "model.layers.32.self_attn.v_proj.weight": "llamaforcausallm__model__layers__32__self_attn__v_proj__weight", "model.layers.32.self_attn.o_proj.weight": "llamaforcausallm__model__layers__32__self_attn__o_proj__weight", "model.layers.32.mlp.gate_proj.weight": "llamaforcausallm__model__layers__32__mlp__gate_proj__weight", "model.layers.32.mlp.up_proj.weight": "llamaforcausallm__model__layers__32__mlp__up_proj__weight", "model.layers.32.mlp.down_proj.weight": "llamaforcausallm__model__layers__32__mlp__down_proj__weight", "model.layers.32.input_layernorm.weight": "llamaforcausallm__model__layers__32__input_layernorm__weight", "model.layers.32.post_attention_layernorm.weight": "llamaforcausallm__model__layers__32__post_attention_layernorm__weight", "model.layers.33.self_attn.q_proj.weight": "llamaforcausallm__model__layers__33__self_attn__q_proj__weight", "model.layers.33.self_attn.k_proj.weight": "llamaforcausallm__model__layers__33__self_attn__k_proj__weight", "model.layers.33.self_attn.v_proj.weight": "llamaforcausallm__model__layers__33__self_attn__v_proj__weight", "model.layers.33.self_attn.o_proj.weight": "llamaforcausallm__model__layers__33__self_attn__o_proj__weight", "model.layers.33.mlp.gate_proj.weight": "llamaforcausallm__model__layers__33__mlp__gate_proj__weight", "model.layers.33.mlp.up_proj.weight": "llamaforcausallm__model__layers__33__mlp__up_proj__weight", "model.layers.33.mlp.down_proj.weight": "llamaforcausallm__model__layers__33__mlp__down_proj__weight", "model.layers.33.input_layernorm.weight": "llamaforcausallm__model__layers__33__input_layernorm__weight", "model.layers.33.post_attention_layernorm.weight": "llamaforcausallm__model__layers__33__post_attention_layernorm__weight", "model.layers.34.self_attn.q_proj.weight": "llamaforcausallm__model__layers__34__self_attn__q_proj__weight", "model.layers.34.self_attn.k_proj.weight": "llamaforcausallm__model__layers__34__self_attn__k_proj__weight", "model.layers.34.self_attn.v_proj.weight": "llamaforcausallm__model__layers__34__self_attn__v_proj__weight", "model.layers.34.self_attn.o_proj.weight": "llamaforcausallm__model__layers__34__self_attn__o_proj__weight", "model.layers.34.mlp.gate_proj.weight": "llamaforcausallm__model__layers__34__mlp__gate_proj__weight", "model.layers.34.mlp.up_proj.weight": "llamaforcausallm__model__layers__34__mlp__up_proj__weight", "model.layers.34.mlp.down_proj.weight": "llamaforcausallm__model__layers__34__mlp__down_proj__weight", "model.layers.34.input_layernorm.weight": "llamaforcausallm__model__layers__34__input_layernorm__weight", "model.layers.34.post_attention_layernorm.weight": "llamaforcausallm__model__layers__34__post_attention_layernorm__weight", "model.layers.35.self_attn.q_proj.weight": "llamaforcausallm__model__layers__35__self_attn__q_proj__weight", "model.layers.35.self_attn.k_proj.weight": "llamaforcausallm__model__layers__35__self_attn__k_proj__weight", "model.layers.35.self_attn.v_proj.weight": "llamaforcausallm__model__layers__35__self_attn__v_proj__weight", "model.layers.35.self_attn.o_proj.weight": "llamaforcausallm__model__layers__35__self_attn__o_proj__weight", "model.layers.35.mlp.gate_proj.weight": "llamaforcausallm__model__layers__35__mlp__gate_proj__weight", "model.layers.35.mlp.up_proj.weight": "llamaforcausallm__model__layers__35__mlp__up_proj__weight", "model.layers.35.mlp.down_proj.weight": "llamaforcausallm__model__layers__35__mlp__down_proj__weight", "model.layers.35.input_layernorm.weight": "llamaforcausallm__model__layers__35__input_layernorm__weight", "model.layers.35.post_attention_layernorm.weight": "llamaforcausallm__model__layers__35__post_attention_layernorm__weight", "model.layers.36.self_attn.q_proj.weight": "llamaforcausallm__model__layers__36__self_attn__q_proj__weight", "model.layers.36.self_attn.k_proj.weight": "llamaforcausallm__model__layers__36__self_attn__k_proj__weight", "model.layers.36.self_attn.v_proj.weight": "llamaforcausallm__model__layers__36__self_attn__v_proj__weight", "model.layers.36.self_attn.o_proj.weight": "llamaforcausallm__model__layers__36__self_attn__o_proj__weight", "model.layers.36.mlp.gate_proj.weight": "llamaforcausallm__model__layers__36__mlp__gate_proj__weight", "model.layers.36.mlp.up_proj.weight": "llamaforcausallm__model__layers__36__mlp__up_proj__weight", "model.layers.36.mlp.down_proj.weight": "llamaforcausallm__model__layers__36__mlp__down_proj__weight", "model.layers.36.input_layernorm.weight": "llamaforcausallm__model__layers__36__input_layernorm__weight", "model.layers.36.post_attention_layernorm.weight": "llamaforcausallm__model__layers__36__post_attention_layernorm__weight", "model.layers.37.self_attn.q_proj.weight": "llamaforcausallm__model__layers__37__self_attn__q_proj__weight", "model.layers.37.self_attn.k_proj.weight": "llamaforcausallm__model__layers__37__self_attn__k_proj__weight", "model.layers.37.self_attn.v_proj.weight": "llamaforcausallm__model__layers__37__self_attn__v_proj__weight", "model.layers.37.self_attn.o_proj.weight": "llamaforcausallm__model__layers__37__self_attn__o_proj__weight", "model.layers.37.mlp.gate_proj.weight": "llamaforcausallm__model__layers__37__mlp__gate_proj__weight", "model.layers.37.mlp.up_proj.weight": "llamaforcausallm__model__layers__37__mlp__up_proj__weight", "model.layers.37.mlp.down_proj.weight": "llamaforcausallm__model__layers__37__mlp__down_proj__weight", "model.layers.37.input_layernorm.weight": "llamaforcausallm__model__layers__37__input_layernorm__weight", "model.layers.37.post_attention_layernorm.weight": "llamaforcausallm__model__layers__37__post_attention_layernorm__weight", "model.layers.38.self_attn.q_proj.weight": "llamaforcausallm__model__layers__38__self_attn__q_proj__weight", "model.layers.38.self_attn.k_proj.weight": "llamaforcausallm__model__layers__38__self_attn__k_proj__weight", "model.layers.38.self_attn.v_proj.weight": "llamaforcausallm__model__layers__38__self_attn__v_proj__weight", "model.layers.38.self_attn.o_proj.weight": "llamaforcausallm__model__layers__38__self_attn__o_proj__weight", "model.layers.38.mlp.gate_proj.weight": "llamaforcausallm__model__layers__38__mlp__gate_proj__weight", "model.layers.38.mlp.up_proj.weight": "llamaforcausallm__model__layers__38__mlp__up_proj__weight", "model.layers.38.mlp.down_proj.weight": "llamaforcausallm__model__layers__38__mlp__down_proj__weight", "model.layers.38.input_layernorm.weight": "llamaforcausallm__model__layers__38__input_layernorm__weight", "model.layers.38.post_attention_layernorm.weight": "llamaforcausallm__model__layers__38__post_attention_layernorm__weight", "model.layers.39.self_attn.q_proj.weight": "llamaforcausallm__model__layers__39__self_attn__q_proj__weight", "model.layers.39.self_attn.k_proj.weight": "llamaforcausallm__model__layers__39__self_attn__k_proj__weight", "model.layers.39.self_attn.v_proj.weight": "llamaforcausallm__model__layers__39__self_attn__v_proj__weight", "model.layers.39.self_attn.o_proj.weight": "llamaforcausallm__model__layers__39__self_attn__o_proj__weight", "model.layers.39.mlp.gate_proj.weight": "llamaforcausallm__model__layers__39__mlp__gate_proj__weight", "model.layers.39.mlp.up_proj.weight": "llamaforcausallm__model__layers__39__mlp__up_proj__weight", "model.layers.39.mlp.down_proj.weight": "llamaforcausallm__model__layers__39__mlp__down_proj__weight", "model.layers.39.input_layernorm.weight": "llamaforcausallm__model__layers__39__input_layernorm__weight", "model.layers.39.post_attention_layernorm.weight": "llamaforcausallm__model__layers__39__post_attention_layernorm__weight", "model.layers.40.self_attn.q_proj.weight": "llamaforcausallm__model__layers__40__self_attn__q_proj__weight", "model.layers.40.self_attn.k_proj.weight": "llamaforcausallm__model__layers__40__self_attn__k_proj__weight", "model.layers.40.self_attn.v_proj.weight": "llamaforcausallm__model__layers__40__self_attn__v_proj__weight", "model.layers.40.self_attn.o_proj.weight": "llamaforcausallm__model__layers__40__self_attn__o_proj__weight", "model.layers.40.mlp.gate_proj.weight": "llamaforcausallm__model__layers__40__mlp__gate_proj__weight", "model.layers.40.mlp.up_proj.weight": "llamaforcausallm__model__layers__40__mlp__up_proj__weight", "model.layers.40.mlp.down_proj.weight": "llamaforcausallm__model__layers__40__mlp__down_proj__weight", "model.layers.40.input_layernorm.weight": "llamaforcausallm__model__layers__40__input_layernorm__weight", "model.layers.40.post_attention_layernorm.weight": "llamaforcausallm__model__layers__40__post_attention_layernorm__weight", "model.layers.41.self_attn.q_proj.weight": "llamaforcausallm__model__layers__41__self_attn__q_proj__weight", "model.layers.41.self_attn.k_proj.weight": "llamaforcausallm__model__layers__41__self_attn__k_proj__weight", "model.layers.41.self_attn.v_proj.weight": "llamaforcausallm__model__layers__41__self_attn__v_proj__weight", "model.layers.41.self_attn.o_proj.weight": "llamaforcausallm__model__layers__41__self_attn__o_proj__weight", "model.layers.41.mlp.gate_proj.weight": "llamaforcausallm__model__layers__41__mlp__gate_proj__weight", "model.layers.41.mlp.up_proj.weight": "llamaforcausallm__model__layers__41__mlp__up_proj__weight", "model.layers.41.mlp.down_proj.weight": "llamaforcausallm__model__layers__41__mlp__down_proj__weight", "model.layers.41.input_layernorm.weight": "llamaforcausallm__model__layers__41__input_layernorm__weight", "model.layers.41.post_attention_layernorm.weight": "llamaforcausallm__model__layers__41__post_attention_layernorm__weight", "model.layers.42.self_attn.q_proj.weight": "llamaforcausallm__model__layers__42__self_attn__q_proj__weight", "model.layers.42.self_attn.k_proj.weight": "llamaforcausallm__model__layers__42__self_attn__k_proj__weight", "model.layers.42.self_attn.v_proj.weight": "llamaforcausallm__model__layers__42__self_attn__v_proj__weight", "model.layers.42.self_attn.o_proj.weight": "llamaforcausallm__model__layers__42__self_attn__o_proj__weight", "model.layers.42.mlp.gate_proj.weight": "llamaforcausallm__model__layers__42__mlp__gate_proj__weight", "model.layers.42.mlp.up_proj.weight": "llamaforcausallm__model__layers__42__mlp__up_proj__weight", "model.layers.42.mlp.down_proj.weight": "llamaforcausallm__model__layers__42__mlp__down_proj__weight", "model.layers.42.input_layernorm.weight": "llamaforcausallm__model__layers__42__input_layernorm__weight", "model.layers.42.post_attention_layernorm.weight": "llamaforcausallm__model__layers__42__post_attention_layernorm__weight", "model.layers.43.self_attn.q_proj.weight": "llamaforcausallm__model__layers__43__self_attn__q_proj__weight", "model.layers.43.self_attn.k_proj.weight": "llamaforcausallm__model__layers__43__self_attn__k_proj__weight", "model.layers.43.self_attn.v_proj.weight": "llamaforcausallm__model__layers__43__self_attn__v_proj__weight", "model.layers.43.self_attn.o_proj.weight": "llamaforcausallm__model__layers__43__self_attn__o_proj__weight", "model.layers.43.mlp.gate_proj.weight": "llamaforcausallm__model__layers__43__mlp__gate_proj__weight", "model.layers.43.mlp.up_proj.weight": "llamaforcausallm__model__layers__43__mlp__up_proj__weight", "model.layers.43.mlp.down_proj.weight": "llamaforcausallm__model__layers__43__mlp__down_proj__weight", "model.layers.43.input_layernorm.weight": "llamaforcausallm__model__layers__43__input_layernorm__weight", "model.layers.43.post_attention_layernorm.weight": "llamaforcausallm__model__layers__43__post_attention_layernorm__weight", "model.layers.44.self_attn.q_proj.weight": "llamaforcausallm__model__layers__44__self_attn__q_proj__weight", "model.layers.44.self_attn.k_proj.weight": "llamaforcausallm__model__layers__44__self_attn__k_proj__weight", "model.layers.44.self_attn.v_proj.weight": "llamaforcausallm__model__layers__44__self_attn__v_proj__weight", "model.layers.44.self_attn.o_proj.weight": "llamaforcausallm__model__layers__44__self_attn__o_proj__weight", "model.layers.44.mlp.gate_proj.weight": "llamaforcausallm__model__layers__44__mlp__gate_proj__weight", "model.layers.44.mlp.up_proj.weight": "llamaforcausallm__model__layers__44__mlp__up_proj__weight", "model.layers.44.mlp.down_proj.weight": "llamaforcausallm__model__layers__44__mlp__down_proj__weight", "model.layers.44.input_layernorm.weight": "llamaforcausallm__model__layers__44__input_layernorm__weight", "model.layers.44.post_attention_layernorm.weight": "llamaforcausallm__model__layers__44__post_attention_layernorm__weight", "model.layers.45.self_attn.q_proj.weight": "llamaforcausallm__model__layers__45__self_attn__q_proj__weight", "model.layers.45.self_attn.k_proj.weight": "llamaforcausallm__model__layers__45__self_attn__k_proj__weight", "model.layers.45.self_attn.v_proj.weight": "llamaforcausallm__model__layers__45__self_attn__v_proj__weight", "model.layers.45.self_attn.o_proj.weight": "llamaforcausallm__model__layers__45__self_attn__o_proj__weight", "model.layers.45.mlp.gate_proj.weight": "llamaforcausallm__model__layers__45__mlp__gate_proj__weight", "model.layers.45.mlp.up_proj.weight": "llamaforcausallm__model__layers__45__mlp__up_proj__weight", "model.layers.45.mlp.down_proj.weight": "llamaforcausallm__model__layers__45__mlp__down_proj__weight", "model.layers.45.input_layernorm.weight": "llamaforcausallm__model__layers__45__input_layernorm__weight", "model.layers.45.post_attention_layernorm.weight": "llamaforcausallm__model__layers__45__post_attention_layernorm__weight", "model.layers.46.self_attn.q_proj.weight": "llamaforcausallm__model__layers__46__self_attn__q_proj__weight", "model.layers.46.self_attn.k_proj.weight": "llamaforcausallm__model__layers__46__self_attn__k_proj__weight", "model.layers.46.self_attn.v_proj.weight": "llamaforcausallm__model__layers__46__self_attn__v_proj__weight", "model.layers.46.self_attn.o_proj.weight": "llamaforcausallm__model__layers__46__self_attn__o_proj__weight", "model.layers.46.mlp.gate_proj.weight": "llamaforcausallm__model__layers__46__mlp__gate_proj__weight", "model.layers.46.mlp.up_proj.weight": "llamaforcausallm__model__layers__46__mlp__up_proj__weight", "model.layers.46.mlp.down_proj.weight": "llamaforcausallm__model__layers__46__mlp__down_proj__weight", "model.layers.46.input_layernorm.weight": "llamaforcausallm__model__layers__46__input_layernorm__weight", "model.layers.46.post_attention_layernorm.weight": "llamaforcausallm__model__layers__46__post_attention_layernorm__weight", "model.layers.47.self_attn.q_proj.weight": "llamaforcausallm__model__layers__47__self_attn__q_proj__weight", "model.layers.47.self_attn.k_proj.weight": "llamaforcausallm__model__layers__47__self_attn__k_proj__weight", "model.layers.47.self_attn.v_proj.weight": "llamaforcausallm__model__layers__47__self_attn__v_proj__weight", "model.layers.47.self_attn.o_proj.weight": "llamaforcausallm__model__layers__47__self_attn__o_proj__weight", "model.layers.47.mlp.gate_proj.weight": "llamaforcausallm__model__layers__47__mlp__gate_proj__weight", "model.layers.47.mlp.up_proj.weight": "llamaforcausallm__model__layers__47__mlp__up_proj__weight", "model.layers.47.mlp.down_proj.weight": "llamaforcausallm__model__layers__47__mlp__down_proj__weight", "model.layers.47.input_layernorm.weight": "llamaforcausallm__model__layers__47__input_layernorm__weight", "model.layers.47.post_attention_layernorm.weight": "llamaforcausallm__model__layers__47__post_attention_layernorm__weight", "model.layers.48.self_attn.q_proj.weight": "llamaforcausallm__model__layers__48__self_attn__q_proj__weight", "model.layers.48.self_attn.k_proj.weight": "llamaforcausallm__model__layers__48__self_attn__k_proj__weight", "model.layers.48.self_attn.v_proj.weight": "llamaforcausallm__model__layers__48__self_attn__v_proj__weight", "model.layers.48.self_attn.o_proj.weight": "llamaforcausallm__model__layers__48__self_attn__o_proj__weight", "model.layers.48.mlp.gate_proj.weight": "llamaforcausallm__model__layers__48__mlp__gate_proj__weight", "model.layers.48.mlp.up_proj.weight": "llamaforcausallm__model__layers__48__mlp__up_proj__weight", "model.layers.48.mlp.down_proj.weight": "llamaforcausallm__model__layers__48__mlp__down_proj__weight", "model.layers.48.input_layernorm.weight": "llamaforcausallm__model__layers__48__input_layernorm__weight", "model.layers.48.post_attention_layernorm.weight": "llamaforcausallm__model__layers__48__post_attention_layernorm__weight", "model.layers.49.self_attn.q_proj.weight": "llamaforcausallm__model__layers__49__self_attn__q_proj__weight", "model.layers.49.self_attn.k_proj.weight": "llamaforcausallm__model__layers__49__self_attn__k_proj__weight", "model.layers.49.self_attn.v_proj.weight": "llamaforcausallm__model__layers__49__self_attn__v_proj__weight", "model.layers.49.self_attn.o_proj.weight": "llamaforcausallm__model__layers__49__self_attn__o_proj__weight", "model.layers.49.mlp.gate_proj.weight": "llamaforcausallm__model__layers__49__mlp__gate_proj__weight", "model.layers.49.mlp.up_proj.weight": "llamaforcausallm__model__layers__49__mlp__up_proj__weight", "model.layers.49.mlp.down_proj.weight": "llamaforcausallm__model__layers__49__mlp__down_proj__weight", "model.layers.49.input_layernorm.weight": "llamaforcausallm__model__layers__49__input_layernorm__weight", "model.layers.49.post_attention_layernorm.weight": "llamaforcausallm__model__layers__49__post_attention_layernorm__weight", "model.layers.50.self_attn.q_proj.weight": "llamaforcausallm__model__layers__50__self_attn__q_proj__weight", "model.layers.50.self_attn.k_proj.weight": "llamaforcausallm__model__layers__50__self_attn__k_proj__weight", "model.layers.50.self_attn.v_proj.weight": "llamaforcausallm__model__layers__50__self_attn__v_proj__weight", "model.layers.50.self_attn.o_proj.weight": "llamaforcausallm__model__layers__50__self_attn__o_proj__weight", "model.layers.50.mlp.gate_proj.weight": "llamaforcausallm__model__layers__50__mlp__gate_proj__weight", "model.layers.50.mlp.up_proj.weight": "llamaforcausallm__model__layers__50__mlp__up_proj__weight", "model.layers.50.mlp.down_proj.weight": "llamaforcausallm__model__layers__50__mlp__down_proj__weight", "model.layers.50.input_layernorm.weight": "llamaforcausallm__model__layers__50__input_layernorm__weight", "model.layers.50.post_attention_layernorm.weight": "llamaforcausallm__model__layers__50__post_attention_layernorm__weight", "model.layers.51.self_attn.q_proj.weight": "llamaforcausallm__model__layers__51__self_attn__q_proj__weight", "model.layers.51.self_attn.k_proj.weight": "llamaforcausallm__model__layers__51__self_attn__k_proj__weight", "model.layers.51.self_attn.v_proj.weight": "llamaforcausallm__model__layers__51__self_attn__v_proj__weight", "model.layers.51.self_attn.o_proj.weight": "llamaforcausallm__model__layers__51__self_attn__o_proj__weight", "model.layers.51.mlp.gate_proj.weight": "llamaforcausallm__model__layers__51__mlp__gate_proj__weight", "model.layers.51.mlp.up_proj.weight": "llamaforcausallm__model__layers__51__mlp__up_proj__weight", "model.layers.51.mlp.down_proj.weight": "llamaforcausallm__model__layers__51__mlp__down_proj__weight", "model.layers.51.input_layernorm.weight": "llamaforcausallm__model__layers__51__input_layernorm__weight", "model.layers.51.post_attention_layernorm.weight": "llamaforcausallm__model__layers__51__post_attention_layernorm__weight", "model.layers.52.self_attn.q_proj.weight": "llamaforcausallm__model__layers__52__self_attn__q_proj__weight", "model.layers.52.self_attn.k_proj.weight": "llamaforcausallm__model__layers__52__self_attn__k_proj__weight", "model.layers.52.self_attn.v_proj.weight": "llamaforcausallm__model__layers__52__self_attn__v_proj__weight", "model.layers.52.self_attn.o_proj.weight": "llamaforcausallm__model__layers__52__self_attn__o_proj__weight", "model.layers.52.mlp.gate_proj.weight": "llamaforcausallm__model__layers__52__mlp__gate_proj__weight", "model.layers.52.mlp.up_proj.weight": "llamaforcausallm__model__layers__52__mlp__up_proj__weight", "model.layers.52.mlp.down_proj.weight": "llamaforcausallm__model__layers__52__mlp__down_proj__weight", "model.layers.52.input_layernorm.weight": "llamaforcausallm__model__layers__52__input_layernorm__weight", "model.layers.52.post_attention_layernorm.weight": "llamaforcausallm__model__layers__52__post_attention_layernorm__weight", "model.layers.53.self_attn.q_proj.weight": "llamaforcausallm__model__layers__53__self_attn__q_proj__weight", "model.layers.53.self_attn.k_proj.weight": "llamaforcausallm__model__layers__53__self_attn__k_proj__weight", "model.layers.53.self_attn.v_proj.weight": "llamaforcausallm__model__layers__53__self_attn__v_proj__weight", "model.layers.53.self_attn.o_proj.weight": "llamaforcausallm__model__layers__53__self_attn__o_proj__weight", "model.layers.53.mlp.gate_proj.weight": "llamaforcausallm__model__layers__53__mlp__gate_proj__weight", "model.layers.53.mlp.up_proj.weight": "llamaforcausallm__model__layers__53__mlp__up_proj__weight", "model.layers.53.mlp.down_proj.weight": "llamaforcausallm__model__layers__53__mlp__down_proj__weight", "model.layers.53.input_layernorm.weight": "llamaforcausallm__model__layers__53__input_layernorm__weight", "model.layers.53.post_attention_layernorm.weight": "llamaforcausallm__model__layers__53__post_attention_layernorm__weight", "model.layers.54.self_attn.q_proj.weight": "llamaforcausallm__model__layers__54__self_attn__q_proj__weight", "model.layers.54.self_attn.k_proj.weight": "llamaforcausallm__model__layers__54__self_attn__k_proj__weight", "model.layers.54.self_attn.v_proj.weight": "llamaforcausallm__model__layers__54__self_attn__v_proj__weight", "model.layers.54.self_attn.o_proj.weight": "llamaforcausallm__model__layers__54__self_attn__o_proj__weight", "model.layers.54.mlp.gate_proj.weight": "llamaforcausallm__model__layers__54__mlp__gate_proj__weight", "model.layers.54.mlp.up_proj.weight": "llamaforcausallm__model__layers__54__mlp__up_proj__weight", "model.layers.54.mlp.down_proj.weight": "llamaforcausallm__model__layers__54__mlp__down_proj__weight", "model.layers.54.input_layernorm.weight": "llamaforcausallm__model__layers__54__input_layernorm__weight", "model.layers.54.post_attention_layernorm.weight": "llamaforcausallm__model__layers__54__post_attention_layernorm__weight", "model.layers.55.self_attn.q_proj.weight": "llamaforcausallm__model__layers__55__self_attn__q_proj__weight", "model.layers.55.self_attn.k_proj.weight": "llamaforcausallm__model__layers__55__self_attn__k_proj__weight", "model.layers.55.self_attn.v_proj.weight": "llamaforcausallm__model__layers__55__self_attn__v_proj__weight", "model.layers.55.self_attn.o_proj.weight": "llamaforcausallm__model__layers__55__self_attn__o_proj__weight", "model.layers.55.mlp.gate_proj.weight": "llamaforcausallm__model__layers__55__mlp__gate_proj__weight", "model.layers.55.mlp.up_proj.weight": "llamaforcausallm__model__layers__55__mlp__up_proj__weight", "model.layers.55.mlp.down_proj.weight": "llamaforcausallm__model__layers__55__mlp__down_proj__weight", "model.layers.55.input_layernorm.weight": "llamaforcausallm__model__layers__55__input_layernorm__weight", "model.layers.55.post_attention_layernorm.weight": "llamaforcausallm__model__layers__55__post_attention_layernorm__weight", "model.layers.56.self_attn.q_proj.weight": "llamaforcausallm__model__layers__56__self_attn__q_proj__weight", "model.layers.56.self_attn.k_proj.weight": "llamaforcausallm__model__layers__56__self_attn__k_proj__weight", "model.layers.56.self_attn.v_proj.weight": "llamaforcausallm__model__layers__56__self_attn__v_proj__weight", "model.layers.56.self_attn.o_proj.weight": "llamaforcausallm__model__layers__56__self_attn__o_proj__weight", "model.layers.56.mlp.gate_proj.weight": "llamaforcausallm__model__layers__56__mlp__gate_proj__weight", "model.layers.56.mlp.up_proj.weight": "llamaforcausallm__model__layers__56__mlp__up_proj__weight", "model.layers.56.mlp.down_proj.weight": "llamaforcausallm__model__layers__56__mlp__down_proj__weight", "model.layers.56.input_layernorm.weight": "llamaforcausallm__model__layers__56__input_layernorm__weight", "model.layers.56.post_attention_layernorm.weight": "llamaforcausallm__model__layers__56__post_attention_layernorm__weight", "model.layers.57.self_attn.q_proj.weight": "llamaforcausallm__model__layers__57__self_attn__q_proj__weight", "model.layers.57.self_attn.k_proj.weight": "llamaforcausallm__model__layers__57__self_attn__k_proj__weight", "model.layers.57.self_attn.v_proj.weight": "llamaforcausallm__model__layers__57__self_attn__v_proj__weight", "model.layers.57.self_attn.o_proj.weight": "llamaforcausallm__model__layers__57__self_attn__o_proj__weight", "model.layers.57.mlp.gate_proj.weight": "llamaforcausallm__model__layers__57__mlp__gate_proj__weight", "model.layers.57.mlp.up_proj.weight": "llamaforcausallm__model__layers__57__mlp__up_proj__weight", "model.layers.57.mlp.down_proj.weight": "llamaforcausallm__model__layers__57__mlp__down_proj__weight", "model.layers.57.input_layernorm.weight": "llamaforcausallm__model__layers__57__input_layernorm__weight", "model.layers.57.post_attention_layernorm.weight": "llamaforcausallm__model__layers__57__post_attention_layernorm__weight", "model.layers.58.self_attn.q_proj.weight": "llamaforcausallm__model__layers__58__self_attn__q_proj__weight", "model.layers.58.self_attn.k_proj.weight": "llamaforcausallm__model__layers__58__self_attn__k_proj__weight", "model.layers.58.self_attn.v_proj.weight": "llamaforcausallm__model__layers__58__self_attn__v_proj__weight", "model.layers.58.self_attn.o_proj.weight": "llamaforcausallm__model__layers__58__self_attn__o_proj__weight", "model.layers.58.mlp.gate_proj.weight": "llamaforcausallm__model__layers__58__mlp__gate_proj__weight", "model.layers.58.mlp.up_proj.weight": "llamaforcausallm__model__layers__58__mlp__up_proj__weight", "model.layers.58.mlp.down_proj.weight": "llamaforcausallm__model__layers__58__mlp__down_proj__weight", "model.layers.58.input_layernorm.weight": "llamaforcausallm__model__layers__58__input_layernorm__weight", "model.layers.58.post_attention_layernorm.weight": "llamaforcausallm__model__layers__58__post_attention_layernorm__weight", "model.layers.59.self_attn.q_proj.weight": "llamaforcausallm__model__layers__59__self_attn__q_proj__weight", "model.layers.59.self_attn.k_proj.weight": "llamaforcausallm__model__layers__59__self_attn__k_proj__weight", "model.layers.59.self_attn.v_proj.weight": "llamaforcausallm__model__layers__59__self_attn__v_proj__weight", "model.layers.59.self_attn.o_proj.weight": "llamaforcausallm__model__layers__59__self_attn__o_proj__weight", "model.layers.59.mlp.gate_proj.weight": "llamaforcausallm__model__layers__59__mlp__gate_proj__weight", "model.layers.59.mlp.up_proj.weight": "llamaforcausallm__model__layers__59__mlp__up_proj__weight", "model.layers.59.mlp.down_proj.weight": "llamaforcausallm__model__layers__59__mlp__down_proj__weight", "model.layers.59.input_layernorm.weight": "llamaforcausallm__model__layers__59__input_layernorm__weight", "model.layers.59.post_attention_layernorm.weight": "llamaforcausallm__model__layers__59__post_attention_layernorm__weight", "model.layers.60.self_attn.q_proj.weight": "llamaforcausallm__model__layers__60__self_attn__q_proj__weight", "model.layers.60.self_attn.k_proj.weight": "llamaforcausallm__model__layers__60__self_attn__k_proj__weight", "model.layers.60.self_attn.v_proj.weight": "llamaforcausallm__model__layers__60__self_attn__v_proj__weight", "model.layers.60.self_attn.o_proj.weight": "llamaforcausallm__model__layers__60__self_attn__o_proj__weight", "model.layers.60.mlp.gate_proj.weight": "llamaforcausallm__model__layers__60__mlp__gate_proj__weight", "model.layers.60.mlp.up_proj.weight": "llamaforcausallm__model__layers__60__mlp__up_proj__weight", "model.layers.60.mlp.down_proj.weight": "llamaforcausallm__model__layers__60__mlp__down_proj__weight", "model.layers.60.input_layernorm.weight": "llamaforcausallm__model__layers__60__input_layernorm__weight", "model.layers.60.post_attention_layernorm.weight": "llamaforcausallm__model__layers__60__post_attention_layernorm__weight", "model.layers.61.self_attn.q_proj.weight": "llamaforcausallm__model__layers__61__self_attn__q_proj__weight", "model.layers.61.self_attn.k_proj.weight": "llamaforcausallm__model__layers__61__self_attn__k_proj__weight", "model.layers.61.self_attn.v_proj.weight": "llamaforcausallm__model__layers__61__self_attn__v_proj__weight", "model.layers.61.self_attn.o_proj.weight": "llamaforcausallm__model__layers__61__self_attn__o_proj__weight", "model.layers.61.mlp.gate_proj.weight": "llamaforcausallm__model__layers__61__mlp__gate_proj__weight", "model.layers.61.mlp.up_proj.weight": "llamaforcausallm__model__layers__61__mlp__up_proj__weight", "model.layers.61.mlp.down_proj.weight": "llamaforcausallm__model__layers__61__mlp__down_proj__weight", "model.layers.61.input_layernorm.weight": "llamaforcausallm__model__layers__61__input_layernorm__weight", "model.layers.61.post_attention_layernorm.weight": "llamaforcausallm__model__layers__61__post_attention_layernorm__weight", "model.layers.62.self_attn.q_proj.weight": "llamaforcausallm__model__layers__62__self_attn__q_proj__weight", "model.layers.62.self_attn.k_proj.weight": "llamaforcausallm__model__layers__62__self_attn__k_proj__weight", "model.layers.62.self_attn.v_proj.weight": "llamaforcausallm__model__layers__62__self_attn__v_proj__weight", "model.layers.62.self_attn.o_proj.weight": "llamaforcausallm__model__layers__62__self_attn__o_proj__weight", "model.layers.62.mlp.gate_proj.weight": "llamaforcausallm__model__layers__62__mlp__gate_proj__weight", "model.layers.62.mlp.up_proj.weight": "llamaforcausallm__model__layers__62__mlp__up_proj__weight", "model.layers.62.mlp.down_proj.weight": "llamaforcausallm__model__layers__62__mlp__down_proj__weight", "model.layers.62.input_layernorm.weight": "llamaforcausallm__model__layers__62__input_layernorm__weight", "model.layers.62.post_attention_layernorm.weight": "llamaforcausallm__model__layers__62__post_attention_layernorm__weight", "model.layers.63.self_attn.q_proj.weight": "llamaforcausallm__model__layers__63__self_attn__q_proj__weight", "model.layers.63.self_attn.k_proj.weight": "llamaforcausallm__model__layers__63__self_attn__k_proj__weight", "model.layers.63.self_attn.v_proj.weight": "llamaforcausallm__model__layers__63__self_attn__v_proj__weight", "model.layers.63.self_attn.o_proj.weight": "llamaforcausallm__model__layers__63__self_attn__o_proj__weight", "model.layers.63.mlp.gate_proj.weight": "llamaforcausallm__model__layers__63__mlp__gate_proj__weight", "model.layers.63.mlp.up_proj.weight": "llamaforcausallm__model__layers__63__mlp__up_proj__weight", "model.layers.63.mlp.down_proj.weight": "llamaforcausallm__model__layers__63__mlp__down_proj__weight", "model.layers.63.input_layernorm.weight": "llamaforcausallm__model__layers__63__input_layernorm__weight", "model.layers.63.post_attention_layernorm.weight": "llamaforcausallm__model__layers__63__post_attention_layernorm__weight", "model.layers.64.self_attn.q_proj.weight": "llamaforcausallm__model__layers__64__self_attn__q_proj__weight", "model.layers.64.self_attn.k_proj.weight": "llamaforcausallm__model__layers__64__self_attn__k_proj__weight", "model.layers.64.self_attn.v_proj.weight": "llamaforcausallm__model__layers__64__self_attn__v_proj__weight", "model.layers.64.self_attn.o_proj.weight": "llamaforcausallm__model__layers__64__self_attn__o_proj__weight", "model.layers.64.mlp.gate_proj.weight": "llamaforcausallm__model__layers__64__mlp__gate_proj__weight", "model.layers.64.mlp.up_proj.weight": "llamaforcausallm__model__layers__64__mlp__up_proj__weight", "model.layers.64.mlp.down_proj.weight": "llamaforcausallm__model__layers__64__mlp__down_proj__weight", "model.layers.64.input_layernorm.weight": "llamaforcausallm__model__layers__64__input_layernorm__weight", "model.layers.64.post_attention_layernorm.weight": "llamaforcausallm__model__layers__64__post_attention_layernorm__weight", "model.layers.65.self_attn.q_proj.weight": "llamaforcausallm__model__layers__65__self_attn__q_proj__weight", "model.layers.65.self_attn.k_proj.weight": "llamaforcausallm__model__layers__65__self_attn__k_proj__weight", "model.layers.65.self_attn.v_proj.weight": "llamaforcausallm__model__layers__65__self_attn__v_proj__weight", "model.layers.65.self_attn.o_proj.weight": "llamaforcausallm__model__layers__65__self_attn__o_proj__weight", "model.layers.65.mlp.gate_proj.weight": "llamaforcausallm__model__layers__65__mlp__gate_proj__weight", "model.layers.65.mlp.up_proj.weight": "llamaforcausallm__model__layers__65__mlp__up_proj__weight", "model.layers.65.mlp.down_proj.weight": "llamaforcausallm__model__layers__65__mlp__down_proj__weight", "model.layers.65.input_layernorm.weight": "llamaforcausallm__model__layers__65__input_layernorm__weight", "model.layers.65.post_attention_layernorm.weight": "llamaforcausallm__model__layers__65__post_attention_layernorm__weight", "model.layers.66.self_attn.q_proj.weight": "llamaforcausallm__model__layers__66__self_attn__q_proj__weight", "model.layers.66.self_attn.k_proj.weight": "llamaforcausallm__model__layers__66__self_attn__k_proj__weight", "model.layers.66.self_attn.v_proj.weight": "llamaforcausallm__model__layers__66__self_attn__v_proj__weight", "model.layers.66.self_attn.o_proj.weight": "llamaforcausallm__model__layers__66__self_attn__o_proj__weight", "model.layers.66.mlp.gate_proj.weight": "llamaforcausallm__model__layers__66__mlp__gate_proj__weight", "model.layers.66.mlp.up_proj.weight": "llamaforcausallm__model__layers__66__mlp__up_proj__weight", "model.layers.66.mlp.down_proj.weight": "llamaforcausallm__model__layers__66__mlp__down_proj__weight", "model.layers.66.input_layernorm.weight": "llamaforcausallm__model__layers__66__input_layernorm__weight", "model.layers.66.post_attention_layernorm.weight": "llamaforcausallm__model__layers__66__post_attention_layernorm__weight", "model.layers.67.self_attn.q_proj.weight": "llamaforcausallm__model__layers__67__self_attn__q_proj__weight", "model.layers.67.self_attn.k_proj.weight": "llamaforcausallm__model__layers__67__self_attn__k_proj__weight", "model.layers.67.self_attn.v_proj.weight": "llamaforcausallm__model__layers__67__self_attn__v_proj__weight", "model.layers.67.self_attn.o_proj.weight": "llamaforcausallm__model__layers__67__self_attn__o_proj__weight", "model.layers.67.mlp.gate_proj.weight": "llamaforcausallm__model__layers__67__mlp__gate_proj__weight", "model.layers.67.mlp.up_proj.weight": "llamaforcausallm__model__layers__67__mlp__up_proj__weight", "model.layers.67.mlp.down_proj.weight": "llamaforcausallm__model__layers__67__mlp__down_proj__weight", "model.layers.67.input_layernorm.weight": "llamaforcausallm__model__layers__67__input_layernorm__weight", "model.layers.67.post_attention_layernorm.weight": "llamaforcausallm__model__layers__67__post_attention_layernorm__weight", "model.layers.68.self_attn.q_proj.weight": "llamaforcausallm__model__layers__68__self_attn__q_proj__weight", "model.layers.68.self_attn.k_proj.weight": "llamaforcausallm__model__layers__68__self_attn__k_proj__weight", "model.layers.68.self_attn.v_proj.weight": "llamaforcausallm__model__layers__68__self_attn__v_proj__weight", "model.layers.68.self_attn.o_proj.weight": "llamaforcausallm__model__layers__68__self_attn__o_proj__weight", "model.layers.68.mlp.gate_proj.weight": "llamaforcausallm__model__layers__68__mlp__gate_proj__weight", "model.layers.68.mlp.up_proj.weight": "llamaforcausallm__model__layers__68__mlp__up_proj__weight", "model.layers.68.mlp.down_proj.weight": "llamaforcausallm__model__layers__68__mlp__down_proj__weight", "model.layers.68.input_layernorm.weight": "llamaforcausallm__model__layers__68__input_layernorm__weight", "model.layers.68.post_attention_layernorm.weight": "llamaforcausallm__model__layers__68__post_attention_layernorm__weight", "model.layers.69.self_attn.q_proj.weight": "llamaforcausallm__model__layers__69__self_attn__q_proj__weight", "model.layers.69.self_attn.k_proj.weight": "llamaforcausallm__model__layers__69__self_attn__k_proj__weight", "model.layers.69.self_attn.v_proj.weight": "llamaforcausallm__model__layers__69__self_attn__v_proj__weight", "model.layers.69.self_attn.o_proj.weight": "llamaforcausallm__model__layers__69__self_attn__o_proj__weight", "model.layers.69.mlp.gate_proj.weight": "llamaforcausallm__model__layers__69__mlp__gate_proj__weight", "model.layers.69.mlp.up_proj.weight": "llamaforcausallm__model__layers__69__mlp__up_proj__weight", "model.layers.69.mlp.down_proj.weight": "llamaforcausallm__model__layers__69__mlp__down_proj__weight", "model.layers.69.input_layernorm.weight": "llamaforcausallm__model__layers__69__input_layernorm__weight", "model.layers.69.post_attention_layernorm.weight": "llamaforcausallm__model__layers__69__post_attention_layernorm__weight", "model.layers.70.self_attn.q_proj.weight": "llamaforcausallm__model__layers__70__self_attn__q_proj__weight", "model.layers.70.self_attn.k_proj.weight": "llamaforcausallm__model__layers__70__self_attn__k_proj__weight", "model.layers.70.self_attn.v_proj.weight": "llamaforcausallm__model__layers__70__self_attn__v_proj__weight", "model.layers.70.self_attn.o_proj.weight": "llamaforcausallm__model__layers__70__self_attn__o_proj__weight", "model.layers.70.mlp.gate_proj.weight": "llamaforcausallm__model__layers__70__mlp__gate_proj__weight", "model.layers.70.mlp.up_proj.weight": "llamaforcausallm__model__layers__70__mlp__up_proj__weight", "model.layers.70.mlp.down_proj.weight": "llamaforcausallm__model__layers__70__mlp__down_proj__weight", "model.layers.70.input_layernorm.weight": "llamaforcausallm__model__layers__70__input_layernorm__weight", "model.layers.70.post_attention_layernorm.weight": "llamaforcausallm__model__layers__70__post_attention_layernorm__weight", "model.layers.71.self_attn.q_proj.weight": "llamaforcausallm__model__layers__71__self_attn__q_proj__weight", "model.layers.71.self_attn.k_proj.weight": "llamaforcausallm__model__layers__71__self_attn__k_proj__weight", "model.layers.71.self_attn.v_proj.weight": "llamaforcausallm__model__layers__71__self_attn__v_proj__weight", "model.layers.71.self_attn.o_proj.weight": "llamaforcausallm__model__layers__71__self_attn__o_proj__weight", "model.layers.71.mlp.gate_proj.weight": "llamaforcausallm__model__layers__71__mlp__gate_proj__weight", "model.layers.71.mlp.up_proj.weight": "llamaforcausallm__model__layers__71__mlp__up_proj__weight", "model.layers.71.mlp.down_proj.weight": "llamaforcausallm__model__layers__71__mlp__down_proj__weight", "model.layers.71.input_layernorm.weight": "llamaforcausallm__model__layers__71__input_layernorm__weight", "model.layers.71.post_attention_layernorm.weight": "llamaforcausallm__model__layers__71__post_attention_layernorm__weight", "model.layers.72.self_attn.q_proj.weight": "llamaforcausallm__model__layers__72__self_attn__q_proj__weight", "model.layers.72.self_attn.k_proj.weight": "llamaforcausallm__model__layers__72__self_attn__k_proj__weight", "model.layers.72.self_attn.v_proj.weight": "llamaforcausallm__model__layers__72__self_attn__v_proj__weight", "model.layers.72.self_attn.o_proj.weight": "llamaforcausallm__model__layers__72__self_attn__o_proj__weight", "model.layers.72.mlp.gate_proj.weight": "llamaforcausallm__model__layers__72__mlp__gate_proj__weight", "model.layers.72.mlp.up_proj.weight": "llamaforcausallm__model__layers__72__mlp__up_proj__weight", "model.layers.72.mlp.down_proj.weight": "llamaforcausallm__model__layers__72__mlp__down_proj__weight", "model.layers.72.input_layernorm.weight": "llamaforcausallm__model__layers__72__input_layernorm__weight", "model.layers.72.post_attention_layernorm.weight": "llamaforcausallm__model__layers__72__post_attention_layernorm__weight", "model.layers.73.self_attn.q_proj.weight": "llamaforcausallm__model__layers__73__self_attn__q_proj__weight", "model.layers.73.self_attn.k_proj.weight": "llamaforcausallm__model__layers__73__self_attn__k_proj__weight", "model.layers.73.self_attn.v_proj.weight": "llamaforcausallm__model__layers__73__self_attn__v_proj__weight", "model.layers.73.self_attn.o_proj.weight": "llamaforcausallm__model__layers__73__self_attn__o_proj__weight", "model.layers.73.mlp.gate_proj.weight": "llamaforcausallm__model__layers__73__mlp__gate_proj__weight", "model.layers.73.mlp.up_proj.weight": "llamaforcausallm__model__layers__73__mlp__up_proj__weight", "model.layers.73.mlp.down_proj.weight": "llamaforcausallm__model__layers__73__mlp__down_proj__weight", "model.layers.73.input_layernorm.weight": "llamaforcausallm__model__layers__73__input_layernorm__weight", "model.layers.73.post_attention_layernorm.weight": "llamaforcausallm__model__layers__73__post_attention_layernorm__weight", "model.layers.74.self_attn.q_proj.weight": "llamaforcausallm__model__layers__74__self_attn__q_proj__weight", "model.layers.74.self_attn.k_proj.weight": "llamaforcausallm__model__layers__74__self_attn__k_proj__weight", "model.layers.74.self_attn.v_proj.weight": "llamaforcausallm__model__layers__74__self_attn__v_proj__weight", "model.layers.74.self_attn.o_proj.weight": "llamaforcausallm__model__layers__74__self_attn__o_proj__weight", "model.layers.74.mlp.gate_proj.weight": "llamaforcausallm__model__layers__74__mlp__gate_proj__weight", "model.layers.74.mlp.up_proj.weight": "llamaforcausallm__model__layers__74__mlp__up_proj__weight", "model.layers.74.mlp.down_proj.weight": "llamaforcausallm__model__layers__74__mlp__down_proj__weight", "model.layers.74.input_layernorm.weight": "llamaforcausallm__model__layers__74__input_layernorm__weight", "model.layers.74.post_attention_layernorm.weight": "llamaforcausallm__model__layers__74__post_attention_layernorm__weight", "model.layers.75.self_attn.q_proj.weight": "llamaforcausallm__model__layers__75__self_attn__q_proj__weight", "model.layers.75.self_attn.k_proj.weight": "llamaforcausallm__model__layers__75__self_attn__k_proj__weight", "model.layers.75.self_attn.v_proj.weight": "llamaforcausallm__model__layers__75__self_attn__v_proj__weight", "model.layers.75.self_attn.o_proj.weight": "llamaforcausallm__model__layers__75__self_attn__o_proj__weight", "model.layers.75.mlp.gate_proj.weight": "llamaforcausallm__model__layers__75__mlp__gate_proj__weight", "model.layers.75.mlp.up_proj.weight": "llamaforcausallm__model__layers__75__mlp__up_proj__weight", "model.layers.75.mlp.down_proj.weight": "llamaforcausallm__model__layers__75__mlp__down_proj__weight", "model.layers.75.input_layernorm.weight": "llamaforcausallm__model__layers__75__input_layernorm__weight", "model.layers.75.post_attention_layernorm.weight": "llamaforcausallm__model__layers__75__post_attention_layernorm__weight", "model.layers.76.self_attn.q_proj.weight": "llamaforcausallm__model__layers__76__self_attn__q_proj__weight", "model.layers.76.self_attn.k_proj.weight": "llamaforcausallm__model__layers__76__self_attn__k_proj__weight", "model.layers.76.self_attn.v_proj.weight": "llamaforcausallm__model__layers__76__self_attn__v_proj__weight", "model.layers.76.self_attn.o_proj.weight": "llamaforcausallm__model__layers__76__self_attn__o_proj__weight", "model.layers.76.mlp.gate_proj.weight": "llamaforcausallm__model__layers__76__mlp__gate_proj__weight", "model.layers.76.mlp.up_proj.weight": "llamaforcausallm__model__layers__76__mlp__up_proj__weight", "model.layers.76.mlp.down_proj.weight": "llamaforcausallm__model__layers__76__mlp__down_proj__weight", "model.layers.76.input_layernorm.weight": "llamaforcausallm__model__layers__76__input_layernorm__weight", "model.layers.76.post_attention_layernorm.weight": "llamaforcausallm__model__layers__76__post_attention_layernorm__weight", "model.layers.77.self_attn.q_proj.weight": "llamaforcausallm__model__layers__77__self_attn__q_proj__weight", "model.layers.77.self_attn.k_proj.weight": "llamaforcausallm__model__layers__77__self_attn__k_proj__weight", "model.layers.77.self_attn.v_proj.weight": "llamaforcausallm__model__layers__77__self_attn__v_proj__weight", "model.layers.77.self_attn.o_proj.weight": "llamaforcausallm__model__layers__77__self_attn__o_proj__weight", "model.layers.77.mlp.gate_proj.weight": "llamaforcausallm__model__layers__77__mlp__gate_proj__weight", "model.layers.77.mlp.up_proj.weight": "llamaforcausallm__model__layers__77__mlp__up_proj__weight", "model.layers.77.mlp.down_proj.weight": "llamaforcausallm__model__layers__77__mlp__down_proj__weight", "model.layers.77.input_layernorm.weight": "llamaforcausallm__model__layers__77__input_layernorm__weight", "model.layers.77.post_attention_layernorm.weight": "llamaforcausallm__model__layers__77__post_attention_layernorm__weight", "model.layers.78.self_attn.q_proj.weight": "llamaforcausallm__model__layers__78__self_attn__q_proj__weight", "model.layers.78.self_attn.k_proj.weight": "llamaforcausallm__model__layers__78__self_attn__k_proj__weight", "model.layers.78.self_attn.v_proj.weight": "llamaforcausallm__model__layers__78__self_attn__v_proj__weight", "model.layers.78.self_attn.o_proj.weight": "llamaforcausallm__model__layers__78__self_attn__o_proj__weight", "model.layers.78.mlp.gate_proj.weight": "llamaforcausallm__model__layers__78__mlp__gate_proj__weight", "model.layers.78.mlp.up_proj.weight": "llamaforcausallm__model__layers__78__mlp__up_proj__weight", "model.layers.78.mlp.down_proj.weight": "llamaforcausallm__model__layers__78__mlp__down_proj__weight", "model.layers.78.input_layernorm.weight": "llamaforcausallm__model__layers__78__input_layernorm__weight", "model.layers.78.post_attention_layernorm.weight": "llamaforcausallm__model__layers__78__post_attention_layernorm__weight", "model.layers.79.self_attn.q_proj.weight": "llamaforcausallm__model__layers__79__self_attn__q_proj__weight", "model.layers.79.self_attn.k_proj.weight": "llamaforcausallm__model__layers__79__self_attn__k_proj__weight", "model.layers.79.self_attn.v_proj.weight": "llamaforcausallm__model__layers__79__self_attn__v_proj__weight", "model.layers.79.self_attn.o_proj.weight": "llamaforcausallm__model__layers__79__self_attn__o_proj__weight", "model.layers.79.mlp.gate_proj.weight": "llamaforcausallm__model__layers__79__mlp__gate_proj__weight", "model.layers.79.mlp.up_proj.weight": "llamaforcausallm__model__layers__79__mlp__up_proj__weight", "model.layers.79.mlp.down_proj.weight": "llamaforcausallm__model__layers__79__mlp__down_proj__weight", "model.layers.79.input_layernorm.weight": "llamaforcausallm__model__layers__79__input_layernorm__weight", "model.layers.79.post_attention_layernorm.weight": "llamaforcausallm__model__layers__79__post_attention_layernorm__weight", "model.norm.weight": "llamaforcausallm__model__norm__weight", "lm_head.weight": "llamaforcausallm__lm_head__weight"}
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"bos_token": {
|
3 |
+
"content": "<s>",
|
4 |
+
"lstrip": false,
|
5 |
+
"normalized": false,
|
6 |
+
"rstrip": false,
|
7 |
+
"single_word": false
|
8 |
+
},
|
9 |
+
"eos_token": {
|
10 |
+
"content": "</s>",
|
11 |
+
"lstrip": false,
|
12 |
+
"normalized": false,
|
13 |
+
"rstrip": false,
|
14 |
+
"single_word": false
|
15 |
+
},
|
16 |
+
"unk_token": {
|
17 |
+
"content": "<unk>",
|
18 |
+
"lstrip": false,
|
19 |
+
"normalized": false,
|
20 |
+
"rstrip": false,
|
21 |
+
"single_word": false
|
22 |
+
}
|
23 |
+
}
|
The diff for this file is too large to render.
See raw diff
|
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:9e556afd44213b6bd1be2b850ebbbd98f5481437a8021afaf58ee7fb1818d347
|
3 |
+
size 499723
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"bos_token": {
|
3 |
+
"__type": "AddedToken",
|
4 |
+
"content": "<s>",
|
5 |
+
"lstrip": false,
|
6 |
+
"normalized": false,
|
7 |
+
"rstrip": false,
|
8 |
+
"single_word": false
|
9 |
+
},
|
10 |
+
"clean_up_tokenization_spaces": false,
|
11 |
+
"eos_token": {
|
12 |
+
"__type": "AddedToken",
|
13 |
+
"content": "</s>",
|
14 |
+
"lstrip": false,
|
15 |
+
"normalized": false,
|
16 |
+
"rstrip": false,
|
17 |
+
"single_word": false
|
18 |
+
},
|
19 |
+
"legacy": false,
|
20 |
+
"model_max_length": 1000000000000000019884624838656,
|
21 |
+
"pad_token": null,
|
22 |
+
"sp_model_kwargs": {},
|
23 |
+
"tokenizer_class": "LlamaTokenizer",
|
24 |
+
"unk_token": {
|
25 |
+
"__type": "AddedToken",
|
26 |
+
"content": "<unk>",
|
27 |
+
"lstrip": false,
|
28 |
+
"normalized": false,
|
29 |
+
"rstrip": false,
|
30 |
+
"single_word": false
|
31 |
+
},
|
32 |
+
"use_default_system_prompt": true
|
33 |
+
}
|