2ecd920c993281f31ff8d80f0f9793073a2c34bbbea057ffea801346cfefe1ab
Browse files- README.md +50 -0
- config.json +38 -0
- huggingface-metadata.txt +56 -0
- model.safetensors.index.json +1 -0
- special_tokens_map.json +23 -0
- tokenizer.json +0 -0
- tokenizer.model +3 -0
- tokenizer_config.json +0 -0
README.md
ADDED
@@ -0,0 +1,50 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
---
|
2 |
+
license: other
|
3 |
+
---
|
4 |
+
# Join our Discord! https://discord.gg/Nbv9pQ88Xb
|
5 |
+
## 1500+ members strong 💪
|
6 |
+
---
|
7 |
+
|
8 |
+
[BeaverAI](https://huggingface.co/BeaverAI) proudly presents...
|
9 |
+
|
10 |
+
# Behemoth 123B v1 🦣
|
11 |
+
|
12 |
+
*When you spend your whole life living under a dome, even the idea of an ocean seems impossible to imagine.*
|
13 |
+
|
14 |
+
![image/png](https://cdn-uploads.huggingface.co/production/uploads/65f2fd1c25b848bd061b5c2e/5405NZoj_ptSMO_qM09EW.png)
|
15 |
+
|
16 |
+
## Description
|
17 |
+
|
18 |
+
Testers have reported:
|
19 |
+
- Better creativity and variety
|
20 |
+
- Improved prose
|
21 |
+
- Less positivity, more unhinged (especially on Metharme)
|
22 |
+
- Good intelligence, sharp on nuances and recall.
|
23 |
+
|
24 |
+
## Links
|
25 |
+
- Original: https://huggingface.co/TheDrummer/Behemoth-123B-v1
|
26 |
+
- GGUF: https://huggingface.co/TheDrummer/Behemoth-123B-v1-GGUF
|
27 |
+
- iMatrix: https://huggingface.co/bartowski/Behemoth-123B-v1-GGUF (recommended for small quants)
|
28 |
+
|
29 |
+
## Arsenal (Supported Chat Templates)
|
30 |
+
- Mistral for Instruct / RP / Story
|
31 |
+
- Smart, adaptable, familiar
|
32 |
+
- Metharme (a.k.a. Pygmalion in ST) for RP / Story
|
33 |
+
- Creative, unhinged, unique
|
34 |
+
- Text Completion for RP
|
35 |
+
- You can mix it up and see which works best for you.
|
36 |
+
|
37 |
+
### Favorite RP Format
|
38 |
+
`*action* Dialogue *thoughts* Dialogue *narration*` in 1st person PoV
|
39 |
+
|
40 |
+
## What's Next?
|
41 |
+
- Looking into v1.1...
|
42 |
+
- Already have plans for a v2!
|
43 |
+
|
44 |
+
## Special Thanks
|
45 |
+
- Thank you to each and everyone who donated in [Ko-Fi](https://ko-fi.com/thedrummer) to make our venture a little bit easier.
|
46 |
+
- KinjiHakari777, Dr. Fjut, Kistara, Pseudo, AlexTheVP, Dakkidaze, EvarinSharath'fe, ONTHEREDTEAM, F, Mariana, Garg, Silva, Grozi, & **Phaelon**
|
47 |
+
|
48 |
+
![image/png](https://cdn-uploads.huggingface.co/production/uploads/65f2fd1c25b848bd061b5c2e/KvyYIIA1zkxQNEdGro007.png)
|
49 |
+
|
50 |
+
<audio controls src="https://cdn-uploads.huggingface.co/production/uploads/65f2fd1c25b848bd061b5c2e/FNWdi0WlH-Xd3fjkGVPpp.mpga"></audio>
|
config.json
ADDED
@@ -0,0 +1,38 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"_name_or_path": "mistralai/Mistral-Large-Instruct-2407",
|
3 |
+
"architectures": [
|
4 |
+
"MistralForCausalLM"
|
5 |
+
],
|
6 |
+
"attention_dropout": 0.0,
|
7 |
+
"bos_token_id": 1,
|
8 |
+
"eos_token_id": 2,
|
9 |
+
"head_dim": 128,
|
10 |
+
"hidden_act": "silu",
|
11 |
+
"hidden_size": 12288,
|
12 |
+
"initializer_range": 0.02,
|
13 |
+
"intermediate_size": 28672,
|
14 |
+
"max_position_embeddings": 131072,
|
15 |
+
"model_type": "mistral",
|
16 |
+
"num_attention_heads": 96,
|
17 |
+
"num_hidden_layers": 88,
|
18 |
+
"num_key_value_heads": 8,
|
19 |
+
"rms_norm_eps": 1e-05,
|
20 |
+
"rope_theta": 1000000.0,
|
21 |
+
"sliding_window": null,
|
22 |
+
"tie_word_embeddings": false,
|
23 |
+
"torch_dtype": "bfloat16",
|
24 |
+
"transformers_version": "4.45.2",
|
25 |
+
"use_cache": true,
|
26 |
+
"vocab_size": 32768,
|
27 |
+
"quantization_config": {
|
28 |
+
"quant_method": "exl2",
|
29 |
+
"version": "0.2.3",
|
30 |
+
"bits": 3.0,
|
31 |
+
"head_bits": 6,
|
32 |
+
"calibration": {
|
33 |
+
"rows": 115,
|
34 |
+
"length": 2048,
|
35 |
+
"dataset": "(default)"
|
36 |
+
}
|
37 |
+
}
|
38 |
+
}
|
huggingface-metadata.txt
ADDED
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
url: https://huggingface.co/TheDrummer/Behemoth-123B-v1
|
2 |
+
branch: main
|
3 |
+
download date: 2024-10-10 12:21:40
|
4 |
+
sha256sum:
|
5 |
+
6b3ca45903c0c8cdc2d38012b0574606bbb86b225a43e303cf969e8478869b08 model-00001-of-00051.safetensors
|
6 |
+
e325faa7f9547ea52d31f23942f683824d67b88cc12baa6711ab14d43df90842 model-00002-of-00051.safetensors
|
7 |
+
8fbbdc567e12bfb9167a23a4f2eb94ae23f42f8f00fa4e384138a84282f63f31 model-00003-of-00051.safetensors
|
8 |
+
a145902b8f9547e7d48c0ecc050fcb01551b88a52d2bb222aedf65e2929a3ba1 model-00004-of-00051.safetensors
|
9 |
+
f33874a55ebb5409e87e71085f2d4e04b54fd980336d01bfb9f435e99a28be10 model-00005-of-00051.safetensors
|
10 |
+
ef7cf29534435aaab168220510466a1928f1768797bb84af7be4766c46015ad7 model-00006-of-00051.safetensors
|
11 |
+
4422f3b7cd4cb83c73455dcf8838331b218f31afa1fa62015e077d6abfea050e model-00007-of-00051.safetensors
|
12 |
+
35e9fecfb6021598b57b8e42d2ff050dec62c0a49b9124ac06db1ccb5e4170f5 model-00008-of-00051.safetensors
|
13 |
+
af6813d26816c5384f3723136e691f38039f742f0c64fa2a9b248dc4bf580860 model-00009-of-00051.safetensors
|
14 |
+
be1f44bef6c0fc9a409e41948c6d20cbe13e1366868ad4686706cb6a7b41bdf3 model-00010-of-00051.safetensors
|
15 |
+
3008d7a24ac6f78da455b4c436a62fae8e629e0f1f9c9dd7aa9693fccbee69c2 model-00011-of-00051.safetensors
|
16 |
+
d3e5dab84b7f1b477ecb4059ee6e67dad92f6af94fe99d5f984ca53fbe1337e4 model-00012-of-00051.safetensors
|
17 |
+
ed2460dfa67cdc862b9fcad6327cf789e23887290fc9fc43f4fc9905f77c1b60 model-00013-of-00051.safetensors
|
18 |
+
6dac84aeb5cdf8bdb81466d344ed36444ad273f9dd957be7f74d51bd75e877cd model-00014-of-00051.safetensors
|
19 |
+
d7ee1c94a6f7e8375a4646ae94691d262f9b2f3f0fca37f985cabde9c6612c2a model-00015-of-00051.safetensors
|
20 |
+
512873db81d56821bc9243fefa11a8c369c9b599a2518c171e9e5051420b5c2b model-00016-of-00051.safetensors
|
21 |
+
7f5324799cdb9f9fac358dae741108ff0d1144934573295b952fe98c9cfcfff6 model-00017-of-00051.safetensors
|
22 |
+
cb01c270fa4cfb119550d42078d1ee1956e17b62d5c478ecb0e541481e7767f0 model-00018-of-00051.safetensors
|
23 |
+
e4f0991e649c9ae647073f8fa27da233da812d3ae328ed064a1b5abbcdefd694 model-00019-of-00051.safetensors
|
24 |
+
4860784baed7d8841c9b3e3abce81d27eaf41262c4685225c13bb552ffc6ef71 model-00020-of-00051.safetensors
|
25 |
+
4f37e4ce734f24fb6417e9218c57bfe0205d814820df3059df2991c4ba8f2b91 model-00021-of-00051.safetensors
|
26 |
+
bf7cea21d3f5834bf8f2d0cb4798e2a73944c1d85dfa67f2fa5b5f80d03261b8 model-00022-of-00051.safetensors
|
27 |
+
ca7100bf3b97a28e7e77ab4e78e9364505635d11ec6a6d271dbcbbc889115068 model-00023-of-00051.safetensors
|
28 |
+
7c3690332951668012a208b8c8391884a3bb37b6483bc0428a66b3b750a24c4c model-00024-of-00051.safetensors
|
29 |
+
f627159b2bf21ed5f8ea7bac0cb52cef912429cad35be1197f90dee7f3421bba model-00025-of-00051.safetensors
|
30 |
+
07ced61816e722efc055830dc71850ef7ab96b48849326ccbc3a6f4ad1142b9d model-00026-of-00051.safetensors
|
31 |
+
60e7e6b90444277953ef9c53004eb4e139753a70e0ea37ea2a35ffa8a4c49d53 model-00027-of-00051.safetensors
|
32 |
+
698f7d84236049c419b04eeeba528c610daa91f442ca8fcfe63e378867c5af32 model-00028-of-00051.safetensors
|
33 |
+
cda3aba3e20451b73e2bf475c78c63fa0c85781eaa0e2d5097f4a2448a18fe4a model-00029-of-00051.safetensors
|
34 |
+
4617890580a7e5d324a70a52ebdb7a3c2ebf2e05921f355159c880ae39604190 model-00030-of-00051.safetensors
|
35 |
+
cce0daefef42c53779a8b563dbb5047931f44a944e15fa30dec8ca2cd02b639f model-00031-of-00051.safetensors
|
36 |
+
45e9cb8a2815eb470ae632bf212ea14085898dad4e6221abbee324d0d9e75829 model-00032-of-00051.safetensors
|
37 |
+
4f56abc9f72cb16dd353c939284d15fb43bc1d3e218ef47bec531aed17bf6030 model-00033-of-00051.safetensors
|
38 |
+
bd088c501a387b0b23fc9f1d784a398652d6fc2f844d31cb18bb7eb3e325dba7 model-00034-of-00051.safetensors
|
39 |
+
a7f65d0d220736ffcc8e910c79f8cef4598c22a2f3d0630dc9ac5ab670c8deb4 model-00035-of-00051.safetensors
|
40 |
+
f2640c3a1943bc2004c50017d9de7445f4c6558a6bcf3747090174176acc052d model-00036-of-00051.safetensors
|
41 |
+
961ae49a5c556a26508030d33e6d9f0a8753a7ba5f42a56668e6731de718165f model-00037-of-00051.safetensors
|
42 |
+
09d4ffe0046006f2a3e494d780c7114249eec614ab466a782297d601057e051a model-00038-of-00051.safetensors
|
43 |
+
309f678951aef84d8599905f3eccfc07ac3a80cc385fd8750a9aac8e55bf2646 model-00039-of-00051.safetensors
|
44 |
+
e3c04902a1e31419d9e2f30f05b4f983795ccc4a4a587b6c7aac0e58933dd7c4 model-00040-of-00051.safetensors
|
45 |
+
83420aa63d0fe2300c6e800f918d6ab08dba2547d7dc1be8085b82b18df389d8 model-00041-of-00051.safetensors
|
46 |
+
38d552346018512da1603e1765439522e8ede724b0bc648003a57ee67bad24da model-00042-of-00051.safetensors
|
47 |
+
256e353c5cbd3ca6552e86fd20834437f737d122635bd52ab3a3becf453e5489 model-00043-of-00051.safetensors
|
48 |
+
cc6bf3023536881b6c873480ec0bbc60a6139cc9651431ef5916710caef1c24a model-00044-of-00051.safetensors
|
49 |
+
3463405cb8795d86a28596858e58bb19ba8a27bbbe0bbeabf68b60dbaab003f5 model-00045-of-00051.safetensors
|
50 |
+
1fae8e65f2ac11374c517984cd778dbcf6287a1ac552d7438416f1af93c083d7 model-00046-of-00051.safetensors
|
51 |
+
6d6297a67b53c55fa83e0532fd6a057daaa33bc52192c5889e9180ed6c5e7bbf model-00047-of-00051.safetensors
|
52 |
+
27140b2ab4740263a98632fb65a0c2df8e90bf30b877715d788fc2bb3c2b8b0b model-00048-of-00051.safetensors
|
53 |
+
d5b7cdbcfc01075d6151c117525eeffdcfeea3cef37ba575a37a524cdb343741 model-00049-of-00051.safetensors
|
54 |
+
297d070a9712227b1a6a7b4f62c42195148f83b086870c93efda9c80e31244fd model-00050-of-00051.safetensors
|
55 |
+
67078343e72afd463b3af0b4a0216969f3ad3bb3b4cb40193b9f6eea423fdca9 model-00051-of-00051.safetensors
|
56 |
+
59f95e28944c062244741268596badc900df86c7f5ded05088d2da22a7379e06 tokenizer.model
|
model.safetensors.index.json
ADDED
@@ -0,0 +1 @@
|
|
|
|
|
1 |
+
{"metadata": {"mergekit_version": "0.0.4.4", "total_size": 245220139008}, "weight_map": {"lm_head.weight": "model-00001-of-00051.safetensors", "model.embed_tokens.weight": "model-00001-of-00051.safetensors", "model.layers.0.input_layernorm.weight": "model-00001-of-00051.safetensors", "model.layers.0.mlp.down_proj.weight": "model-00001-of-00051.safetensors", "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00051.safetensors", "model.layers.0.mlp.up_proj.weight": "model-00001-of-00051.safetensors", "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00051.safetensors", "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00051.safetensors", "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00051.safetensors", "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00051.safetensors", "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00051.safetensors", "model.layers.1.input_layernorm.weight": "model-00001-of-00051.safetensors", "model.layers.1.mlp.down_proj.weight": "model-00002-of-00051.safetensors", "model.layers.1.mlp.gate_proj.weight": "model-00002-of-00051.safetensors", "model.layers.1.mlp.up_proj.weight": "model-00002-of-00051.safetensors", "model.layers.1.post_attention_layernorm.weight": "model-00002-of-00051.safetensors", "model.layers.1.self_attn.k_proj.weight": "model-00002-of-00051.safetensors", "model.layers.1.self_attn.o_proj.weight": "model-00002-of-00051.safetensors", "model.layers.1.self_attn.q_proj.weight": "model-00002-of-00051.safetensors", "model.layers.1.self_attn.v_proj.weight": "model-00002-of-00051.safetensors", "model.layers.10.input_layernorm.weight": "model-00002-of-00051.safetensors", "model.layers.10.mlp.down_proj.weight": "model-00002-of-00051.safetensors", "model.layers.10.mlp.gate_proj.weight": "model-00002-of-00051.safetensors", "model.layers.10.mlp.up_proj.weight": "model-00002-of-00051.safetensors", "model.layers.10.post_attention_layernorm.weight": "model-00002-of-00051.safetensors", "model.layers.10.self_attn.k_proj.weight": "model-00002-of-00051.safetensors", "model.layers.10.self_attn.o_proj.weight": "model-00003-of-00051.safetensors", "model.layers.10.self_attn.q_proj.weight": "model-00003-of-00051.safetensors", "model.layers.10.self_attn.v_proj.weight": "model-00003-of-00051.safetensors", "model.layers.11.input_layernorm.weight": "model-00003-of-00051.safetensors", "model.layers.11.mlp.down_proj.weight": "model-00003-of-00051.safetensors", "model.layers.11.mlp.gate_proj.weight": "model-00003-of-00051.safetensors", "model.layers.11.mlp.up_proj.weight": "model-00003-of-00051.safetensors", "model.layers.11.post_attention_layernorm.weight": "model-00003-of-00051.safetensors", "model.layers.11.self_attn.k_proj.weight": "model-00003-of-00051.safetensors", "model.layers.11.self_attn.o_proj.weight": "model-00003-of-00051.safetensors", "model.layers.11.self_attn.q_proj.weight": "model-00003-of-00051.safetensors", "model.layers.11.self_attn.v_proj.weight": "model-00003-of-00051.safetensors", "model.layers.12.input_layernorm.weight": "model-00003-of-00051.safetensors", "model.layers.12.mlp.down_proj.weight": "model-00003-of-00051.safetensors", "model.layers.12.mlp.gate_proj.weight": "model-00003-of-00051.safetensors", "model.layers.12.mlp.up_proj.weight": "model-00004-of-00051.safetensors", "model.layers.12.post_attention_layernorm.weight": "model-00004-of-00051.safetensors", "model.layers.12.self_attn.k_proj.weight": "model-00004-of-00051.safetensors", "model.layers.12.self_attn.o_proj.weight": "model-00004-of-00051.safetensors", "model.layers.12.self_attn.q_proj.weight": "model-00004-of-00051.safetensors", "model.layers.12.self_attn.v_proj.weight": "model-00004-of-00051.safetensors", "model.layers.13.input_layernorm.weight": "model-00004-of-00051.safetensors", "model.layers.13.mlp.down_proj.weight": "model-00004-of-00051.safetensors", "model.layers.13.mlp.gate_proj.weight": "model-00004-of-00051.safetensors", "model.layers.13.mlp.up_proj.weight": "model-00004-of-00051.safetensors", "model.layers.13.post_attention_layernorm.weight": "model-00004-of-00051.safetensors", "model.layers.13.self_attn.k_proj.weight": "model-00004-of-00051.safetensors", "model.layers.13.self_attn.o_proj.weight": "model-00004-of-00051.safetensors", "model.layers.13.self_attn.q_proj.weight": "model-00004-of-00051.safetensors", "model.layers.13.self_attn.v_proj.weight": "model-00004-of-00051.safetensors", "model.layers.14.input_layernorm.weight": "model-00004-of-00051.safetensors", "model.layers.14.mlp.down_proj.weight": "model-00004-of-00051.safetensors", "model.layers.14.mlp.gate_proj.weight": "model-00005-of-00051.safetensors", "model.layers.14.mlp.up_proj.weight": "model-00005-of-00051.safetensors", "model.layers.14.post_attention_layernorm.weight": "model-00005-of-00051.safetensors", "model.layers.14.self_attn.k_proj.weight": "model-00005-of-00051.safetensors", "model.layers.14.self_attn.o_proj.weight": "model-00005-of-00051.safetensors", "model.layers.14.self_attn.q_proj.weight": "model-00005-of-00051.safetensors", "model.layers.14.self_attn.v_proj.weight": "model-00005-of-00051.safetensors", "model.layers.15.input_layernorm.weight": "model-00005-of-00051.safetensors", "model.layers.15.mlp.down_proj.weight": "model-00005-of-00051.safetensors", "model.layers.15.mlp.gate_proj.weight": "model-00005-of-00051.safetensors", "model.layers.15.mlp.up_proj.weight": "model-00005-of-00051.safetensors", "model.layers.15.post_attention_layernorm.weight": "model-00005-of-00051.safetensors", "model.layers.15.self_attn.k_proj.weight": "model-00005-of-00051.safetensors", "model.layers.15.self_attn.o_proj.weight": "model-00005-of-00051.safetensors", "model.layers.15.self_attn.q_proj.weight": "model-00005-of-00051.safetensors", "model.layers.15.self_attn.v_proj.weight": "model-00005-of-00051.safetensors", "model.layers.16.input_layernorm.weight": "model-00005-of-00051.safetensors", "model.layers.16.mlp.down_proj.weight": "model-00006-of-00051.safetensors", "model.layers.16.mlp.gate_proj.weight": "model-00006-of-00051.safetensors", "model.layers.16.mlp.up_proj.weight": "model-00006-of-00051.safetensors", "model.layers.16.post_attention_layernorm.weight": "model-00006-of-00051.safetensors", "model.layers.16.self_attn.k_proj.weight": "model-00006-of-00051.safetensors", "model.layers.16.self_attn.o_proj.weight": "model-00006-of-00051.safetensors", "model.layers.16.self_attn.q_proj.weight": "model-00006-of-00051.safetensors", "model.layers.16.self_attn.v_proj.weight": "model-00006-of-00051.safetensors", "model.layers.17.input_layernorm.weight": "model-00006-of-00051.safetensors", "model.layers.17.mlp.down_proj.weight": "model-00006-of-00051.safetensors", "model.layers.17.mlp.gate_proj.weight": "model-00006-of-00051.safetensors", "model.layers.17.mlp.up_proj.weight": "model-00006-of-00051.safetensors", "model.layers.17.post_attention_layernorm.weight": "model-00006-of-00051.safetensors", "model.layers.17.self_attn.k_proj.weight": "model-00006-of-00051.safetensors", "model.layers.17.self_attn.o_proj.weight": "model-00007-of-00051.safetensors", "model.layers.17.self_attn.q_proj.weight": "model-00007-of-00051.safetensors", "model.layers.17.self_attn.v_proj.weight": "model-00007-of-00051.safetensors", "model.layers.18.input_layernorm.weight": "model-00007-of-00051.safetensors", "model.layers.18.mlp.down_proj.weight": "model-00007-of-00051.safetensors", "model.layers.18.mlp.gate_proj.weight": "model-00007-of-00051.safetensors", "model.layers.18.mlp.up_proj.weight": "model-00007-of-00051.safetensors", "model.layers.18.post_attention_layernorm.weight": "model-00007-of-00051.safetensors", "model.layers.18.self_attn.k_proj.weight": "model-00007-of-00051.safetensors", "model.layers.18.self_attn.o_proj.weight": "model-00007-of-00051.safetensors", "model.layers.18.self_attn.q_proj.weight": "model-00007-of-00051.safetensors", "model.layers.18.self_attn.v_proj.weight": "model-00007-of-00051.safetensors", "model.layers.19.input_layernorm.weight": "model-00007-of-00051.safetensors", "model.layers.19.mlp.down_proj.weight": "model-00007-of-00051.safetensors", "model.layers.19.mlp.gate_proj.weight": "model-00007-of-00051.safetensors", "model.layers.19.mlp.up_proj.weight": "model-00008-of-00051.safetensors", "model.layers.19.post_attention_layernorm.weight": "model-00008-of-00051.safetensors", "model.layers.19.self_attn.k_proj.weight": "model-00008-of-00051.safetensors", "model.layers.19.self_attn.o_proj.weight": "model-00008-of-00051.safetensors", "model.layers.19.self_attn.q_proj.weight": "model-00008-of-00051.safetensors", "model.layers.19.self_attn.v_proj.weight": "model-00008-of-00051.safetensors", "model.layers.2.input_layernorm.weight": "model-00008-of-00051.safetensors", "model.layers.2.mlp.down_proj.weight": "model-00008-of-00051.safetensors", "model.layers.2.mlp.gate_proj.weight": "model-00008-of-00051.safetensors", "model.layers.2.mlp.up_proj.weight": "model-00008-of-00051.safetensors", "model.layers.2.post_attention_layernorm.weight": "model-00008-of-00051.safetensors", "model.layers.2.self_attn.k_proj.weight": "model-00008-of-00051.safetensors", "model.layers.2.self_attn.o_proj.weight": "model-00008-of-00051.safetensors", "model.layers.2.self_attn.q_proj.weight": "model-00008-of-00051.safetensors", "model.layers.2.self_attn.v_proj.weight": "model-00008-of-00051.safetensors", "model.layers.20.input_layernorm.weight": "model-00008-of-00051.safetensors", "model.layers.20.mlp.down_proj.weight": "model-00008-of-00051.safetensors", "model.layers.20.mlp.gate_proj.weight": "model-00009-of-00051.safetensors", "model.layers.20.mlp.up_proj.weight": "model-00009-of-00051.safetensors", "model.layers.20.post_attention_layernorm.weight": "model-00009-of-00051.safetensors", "model.layers.20.self_attn.k_proj.weight": "model-00009-of-00051.safetensors", "model.layers.20.self_attn.o_proj.weight": "model-00009-of-00051.safetensors", "model.layers.20.self_attn.q_proj.weight": "model-00009-of-00051.safetensors", "model.layers.20.self_attn.v_proj.weight": "model-00009-of-00051.safetensors", "model.layers.21.input_layernorm.weight": "model-00009-of-00051.safetensors", "model.layers.21.mlp.down_proj.weight": "model-00009-of-00051.safetensors", "model.layers.21.mlp.gate_proj.weight": "model-00009-of-00051.safetensors", "model.layers.21.mlp.up_proj.weight": "model-00009-of-00051.safetensors", "model.layers.21.post_attention_layernorm.weight": "model-00009-of-00051.safetensors", "model.layers.21.self_attn.k_proj.weight": "model-00009-of-00051.safetensors", "model.layers.21.self_attn.o_proj.weight": "model-00009-of-00051.safetensors", "model.layers.21.self_attn.q_proj.weight": "model-00009-of-00051.safetensors", "model.layers.21.self_attn.v_proj.weight": "model-00009-of-00051.safetensors", "model.layers.22.input_layernorm.weight": "model-00009-of-00051.safetensors", "model.layers.22.mlp.down_proj.weight": "model-00010-of-00051.safetensors", "model.layers.22.mlp.gate_proj.weight": "model-00010-of-00051.safetensors", "model.layers.22.mlp.up_proj.weight": "model-00010-of-00051.safetensors", "model.layers.22.post_attention_layernorm.weight": "model-00010-of-00051.safetensors", "model.layers.22.self_attn.k_proj.weight": "model-00010-of-00051.safetensors", "model.layers.22.self_attn.o_proj.weight": "model-00010-of-00051.safetensors", "model.layers.22.self_attn.q_proj.weight": "model-00010-of-00051.safetensors", "model.layers.22.self_attn.v_proj.weight": "model-00010-of-00051.safetensors", "model.layers.23.input_layernorm.weight": "model-00010-of-00051.safetensors", "model.layers.23.mlp.down_proj.weight": "model-00010-of-00051.safetensors", "model.layers.23.mlp.gate_proj.weight": "model-00010-of-00051.safetensors", "model.layers.23.mlp.up_proj.weight": "model-00010-of-00051.safetensors", "model.layers.23.post_attention_layernorm.weight": "model-00010-of-00051.safetensors", "model.layers.23.self_attn.k_proj.weight": "model-00010-of-00051.safetensors", "model.layers.23.self_attn.o_proj.weight": "model-00011-of-00051.safetensors", "model.layers.23.self_attn.q_proj.weight": "model-00011-of-00051.safetensors", "model.layers.23.self_attn.v_proj.weight": "model-00011-of-00051.safetensors", "model.layers.24.input_layernorm.weight": "model-00011-of-00051.safetensors", "model.layers.24.mlp.down_proj.weight": "model-00011-of-00051.safetensors", "model.layers.24.mlp.gate_proj.weight": "model-00011-of-00051.safetensors", "model.layers.24.mlp.up_proj.weight": "model-00011-of-00051.safetensors", "model.layers.24.post_attention_layernorm.weight": "model-00011-of-00051.safetensors", "model.layers.24.self_attn.k_proj.weight": "model-00011-of-00051.safetensors", "model.layers.24.self_attn.o_proj.weight": "model-00011-of-00051.safetensors", "model.layers.24.self_attn.q_proj.weight": "model-00011-of-00051.safetensors", "model.layers.24.self_attn.v_proj.weight": "model-00011-of-00051.safetensors", "model.layers.25.input_layernorm.weight": "model-00011-of-00051.safetensors", "model.layers.25.mlp.down_proj.weight": "model-00011-of-00051.safetensors", "model.layers.25.mlp.gate_proj.weight": "model-00011-of-00051.safetensors", "model.layers.25.mlp.up_proj.weight": "model-00012-of-00051.safetensors", "model.layers.25.post_attention_layernorm.weight": "model-00012-of-00051.safetensors", "model.layers.25.self_attn.k_proj.weight": "model-00012-of-00051.safetensors", "model.layers.25.self_attn.o_proj.weight": "model-00012-of-00051.safetensors", "model.layers.25.self_attn.q_proj.weight": "model-00012-of-00051.safetensors", "model.layers.25.self_attn.v_proj.weight": "model-00012-of-00051.safetensors", "model.layers.26.input_layernorm.weight": "model-00012-of-00051.safetensors", "model.layers.26.mlp.down_proj.weight": "model-00012-of-00051.safetensors", "model.layers.26.mlp.gate_proj.weight": "model-00012-of-00051.safetensors", "model.layers.26.mlp.up_proj.weight": "model-00012-of-00051.safetensors", "model.layers.26.post_attention_layernorm.weight": "model-00012-of-00051.safetensors", "model.layers.26.self_attn.k_proj.weight": "model-00012-of-00051.safetensors", "model.layers.26.self_attn.o_proj.weight": "model-00012-of-00051.safetensors", "model.layers.26.self_attn.q_proj.weight": "model-00012-of-00051.safetensors", "model.layers.26.self_attn.v_proj.weight": "model-00012-of-00051.safetensors", "model.layers.27.input_layernorm.weight": "model-00012-of-00051.safetensors", "model.layers.27.mlp.down_proj.weight": "model-00012-of-00051.safetensors", "model.layers.27.mlp.gate_proj.weight": "model-00013-of-00051.safetensors", "model.layers.27.mlp.up_proj.weight": "model-00013-of-00051.safetensors", "model.layers.27.post_attention_layernorm.weight": "model-00013-of-00051.safetensors", "model.layers.27.self_attn.k_proj.weight": "model-00013-of-00051.safetensors", "model.layers.27.self_attn.o_proj.weight": "model-00013-of-00051.safetensors", "model.layers.27.self_attn.q_proj.weight": "model-00013-of-00051.safetensors", "model.layers.27.self_attn.v_proj.weight": "model-00013-of-00051.safetensors", "model.layers.28.input_layernorm.weight": "model-00013-of-00051.safetensors", "model.layers.28.mlp.down_proj.weight": "model-00013-of-00051.safetensors", "model.layers.28.mlp.gate_proj.weight": "model-00013-of-00051.safetensors", "model.layers.28.mlp.up_proj.weight": "model-00013-of-00051.safetensors", "model.layers.28.post_attention_layernorm.weight": "model-00013-of-00051.safetensors", "model.layers.28.self_attn.k_proj.weight": "model-00013-of-00051.safetensors", "model.layers.28.self_attn.o_proj.weight": "model-00013-of-00051.safetensors", "model.layers.28.self_attn.q_proj.weight": "model-00013-of-00051.safetensors", "model.layers.28.self_attn.v_proj.weight": "model-00013-of-00051.safetensors", "model.layers.29.input_layernorm.weight": "model-00013-of-00051.safetensors", "model.layers.29.mlp.down_proj.weight": "model-00014-of-00051.safetensors", "model.layers.29.mlp.gate_proj.weight": "model-00014-of-00051.safetensors", "model.layers.29.mlp.up_proj.weight": "model-00014-of-00051.safetensors", "model.layers.29.post_attention_layernorm.weight": "model-00014-of-00051.safetensors", "model.layers.29.self_attn.k_proj.weight": "model-00014-of-00051.safetensors", "model.layers.29.self_attn.o_proj.weight": "model-00014-of-00051.safetensors", "model.layers.29.self_attn.q_proj.weight": "model-00014-of-00051.safetensors", "model.layers.29.self_attn.v_proj.weight": "model-00014-of-00051.safetensors", "model.layers.3.input_layernorm.weight": "model-00014-of-00051.safetensors", "model.layers.3.mlp.down_proj.weight": "model-00014-of-00051.safetensors", "model.layers.3.mlp.gate_proj.weight": "model-00014-of-00051.safetensors", "model.layers.3.mlp.up_proj.weight": "model-00014-of-00051.safetensors", "model.layers.3.post_attention_layernorm.weight": "model-00014-of-00051.safetensors", "model.layers.3.self_attn.k_proj.weight": "model-00014-of-00051.safetensors", "model.layers.3.self_attn.o_proj.weight": "model-00015-of-00051.safetensors", "model.layers.3.self_attn.q_proj.weight": "model-00015-of-00051.safetensors", "model.layers.3.self_attn.v_proj.weight": "model-00015-of-00051.safetensors", "model.layers.30.input_layernorm.weight": "model-00015-of-00051.safetensors", "model.layers.30.mlp.down_proj.weight": "model-00015-of-00051.safetensors", "model.layers.30.mlp.gate_proj.weight": "model-00015-of-00051.safetensors", "model.layers.30.mlp.up_proj.weight": "model-00015-of-00051.safetensors", "model.layers.30.post_attention_layernorm.weight": "model-00015-of-00051.safetensors", "model.layers.30.self_attn.k_proj.weight": "model-00015-of-00051.safetensors", "model.layers.30.self_attn.o_proj.weight": "model-00015-of-00051.safetensors", "model.layers.30.self_attn.q_proj.weight": "model-00015-of-00051.safetensors", "model.layers.30.self_attn.v_proj.weight": "model-00015-of-00051.safetensors", "model.layers.31.input_layernorm.weight": "model-00015-of-00051.safetensors", "model.layers.31.mlp.down_proj.weight": "model-00015-of-00051.safetensors", "model.layers.31.mlp.gate_proj.weight": "model-00015-of-00051.safetensors", "model.layers.31.mlp.up_proj.weight": "model-00016-of-00051.safetensors", "model.layers.31.post_attention_layernorm.weight": "model-00016-of-00051.safetensors", "model.layers.31.self_attn.k_proj.weight": "model-00016-of-00051.safetensors", "model.layers.31.self_attn.o_proj.weight": "model-00016-of-00051.safetensors", "model.layers.31.self_attn.q_proj.weight": "model-00016-of-00051.safetensors", "model.layers.31.self_attn.v_proj.weight": "model-00016-of-00051.safetensors", "model.layers.32.input_layernorm.weight": "model-00016-of-00051.safetensors", "model.layers.32.mlp.down_proj.weight": "model-00016-of-00051.safetensors", "model.layers.32.mlp.gate_proj.weight": "model-00016-of-00051.safetensors", "model.layers.32.mlp.up_proj.weight": "model-00016-of-00051.safetensors", "model.layers.32.post_attention_layernorm.weight": "model-00016-of-00051.safetensors", "model.layers.32.self_attn.k_proj.weight": "model-00016-of-00051.safetensors", "model.layers.32.self_attn.o_proj.weight": "model-00016-of-00051.safetensors", "model.layers.32.self_attn.q_proj.weight": "model-00016-of-00051.safetensors", "model.layers.32.self_attn.v_proj.weight": "model-00016-of-00051.safetensors", "model.layers.33.input_layernorm.weight": "model-00016-of-00051.safetensors", "model.layers.33.mlp.down_proj.weight": "model-00016-of-00051.safetensors", "model.layers.33.mlp.gate_proj.weight": "model-00017-of-00051.safetensors", "model.layers.33.mlp.up_proj.weight": "model-00017-of-00051.safetensors", "model.layers.33.post_attention_layernorm.weight": "model-00017-of-00051.safetensors", "model.layers.33.self_attn.k_proj.weight": "model-00017-of-00051.safetensors", "model.layers.33.self_attn.o_proj.weight": "model-00017-of-00051.safetensors", "model.layers.33.self_attn.q_proj.weight": "model-00017-of-00051.safetensors", "model.layers.33.self_attn.v_proj.weight": "model-00017-of-00051.safetensors", "model.layers.34.input_layernorm.weight": "model-00017-of-00051.safetensors", "model.layers.34.mlp.down_proj.weight": "model-00017-of-00051.safetensors", "model.layers.34.mlp.gate_proj.weight": "model-00017-of-00051.safetensors", "model.layers.34.mlp.up_proj.weight": "model-00017-of-00051.safetensors", "model.layers.34.post_attention_layernorm.weight": "model-00017-of-00051.safetensors", "model.layers.34.self_attn.k_proj.weight": "model-00017-of-00051.safetensors", "model.layers.34.self_attn.o_proj.weight": "model-00017-of-00051.safetensors", "model.layers.34.self_attn.q_proj.weight": "model-00017-of-00051.safetensors", "model.layers.34.self_attn.v_proj.weight": "model-00017-of-00051.safetensors", "model.layers.35.input_layernorm.weight": "model-00017-of-00051.safetensors", "model.layers.35.mlp.down_proj.weight": "model-00018-of-00051.safetensors", "model.layers.35.mlp.gate_proj.weight": "model-00018-of-00051.safetensors", "model.layers.35.mlp.up_proj.weight": "model-00018-of-00051.safetensors", "model.layers.35.post_attention_layernorm.weight": "model-00018-of-00051.safetensors", "model.layers.35.self_attn.k_proj.weight": "model-00018-of-00051.safetensors", "model.layers.35.self_attn.o_proj.weight": "model-00018-of-00051.safetensors", "model.layers.35.self_attn.q_proj.weight": "model-00018-of-00051.safetensors", "model.layers.35.self_attn.v_proj.weight": "model-00018-of-00051.safetensors", "model.layers.36.input_layernorm.weight": "model-00018-of-00051.safetensors", "model.layers.36.mlp.down_proj.weight": "model-00018-of-00051.safetensors", "model.layers.36.mlp.gate_proj.weight": "model-00018-of-00051.safetensors", "model.layers.36.mlp.up_proj.weight": "model-00018-of-00051.safetensors", "model.layers.36.post_attention_layernorm.weight": "model-00018-of-00051.safetensors", "model.layers.36.self_attn.k_proj.weight": "model-00018-of-00051.safetensors", "model.layers.36.self_attn.o_proj.weight": "model-00019-of-00051.safetensors", "model.layers.36.self_attn.q_proj.weight": "model-00019-of-00051.safetensors", "model.layers.36.self_attn.v_proj.weight": "model-00019-of-00051.safetensors", "model.layers.37.input_layernorm.weight": "model-00019-of-00051.safetensors", "model.layers.37.mlp.down_proj.weight": "model-00019-of-00051.safetensors", "model.layers.37.mlp.gate_proj.weight": "model-00019-of-00051.safetensors", "model.layers.37.mlp.up_proj.weight": "model-00019-of-00051.safetensors", "model.layers.37.post_attention_layernorm.weight": "model-00019-of-00051.safetensors", "model.layers.37.self_attn.k_proj.weight": "model-00019-of-00051.safetensors", "model.layers.37.self_attn.o_proj.weight": "model-00019-of-00051.safetensors", "model.layers.37.self_attn.q_proj.weight": "model-00019-of-00051.safetensors", "model.layers.37.self_attn.v_proj.weight": "model-00019-of-00051.safetensors", "model.layers.38.input_layernorm.weight": "model-00019-of-00051.safetensors", "model.layers.38.mlp.down_proj.weight": "model-00019-of-00051.safetensors", "model.layers.38.mlp.gate_proj.weight": "model-00019-of-00051.safetensors", "model.layers.38.mlp.up_proj.weight": "model-00020-of-00051.safetensors", "model.layers.38.post_attention_layernorm.weight": "model-00020-of-00051.safetensors", "model.layers.38.self_attn.k_proj.weight": "model-00020-of-00051.safetensors", "model.layers.38.self_attn.o_proj.weight": "model-00020-of-00051.safetensors", "model.layers.38.self_attn.q_proj.weight": "model-00020-of-00051.safetensors", "model.layers.38.self_attn.v_proj.weight": "model-00020-of-00051.safetensors", "model.layers.39.input_layernorm.weight": "model-00020-of-00051.safetensors", "model.layers.39.mlp.down_proj.weight": "model-00020-of-00051.safetensors", "model.layers.39.mlp.gate_proj.weight": "model-00020-of-00051.safetensors", "model.layers.39.mlp.up_proj.weight": "model-00020-of-00051.safetensors", "model.layers.39.post_attention_layernorm.weight": "model-00020-of-00051.safetensors", "model.layers.39.self_attn.k_proj.weight": "model-00020-of-00051.safetensors", "model.layers.39.self_attn.o_proj.weight": "model-00020-of-00051.safetensors", "model.layers.39.self_attn.q_proj.weight": "model-00020-of-00051.safetensors", "model.layers.39.self_attn.v_proj.weight": "model-00020-of-00051.safetensors", "model.layers.4.input_layernorm.weight": "model-00020-of-00051.safetensors", "model.layers.4.mlp.down_proj.weight": "model-00020-of-00051.safetensors", "model.layers.4.mlp.gate_proj.weight": "model-00021-of-00051.safetensors", "model.layers.4.mlp.up_proj.weight": "model-00021-of-00051.safetensors", "model.layers.4.post_attention_layernorm.weight": "model-00021-of-00051.safetensors", "model.layers.4.self_attn.k_proj.weight": "model-00021-of-00051.safetensors", "model.layers.4.self_attn.o_proj.weight": "model-00021-of-00051.safetensors", "model.layers.4.self_attn.q_proj.weight": "model-00021-of-00051.safetensors", "model.layers.4.self_attn.v_proj.weight": "model-00021-of-00051.safetensors", "model.layers.40.input_layernorm.weight": "model-00021-of-00051.safetensors", "model.layers.40.mlp.down_proj.weight": "model-00021-of-00051.safetensors", "model.layers.40.mlp.gate_proj.weight": "model-00021-of-00051.safetensors", "model.layers.40.mlp.up_proj.weight": "model-00021-of-00051.safetensors", "model.layers.40.post_attention_layernorm.weight": "model-00021-of-00051.safetensors", "model.layers.40.self_attn.k_proj.weight": "model-00021-of-00051.safetensors", "model.layers.40.self_attn.o_proj.weight": "model-00021-of-00051.safetensors", "model.layers.40.self_attn.q_proj.weight": "model-00021-of-00051.safetensors", "model.layers.40.self_attn.v_proj.weight": "model-00021-of-00051.safetensors", "model.layers.41.input_layernorm.weight": "model-00021-of-00051.safetensors", "model.layers.41.mlp.down_proj.weight": "model-00022-of-00051.safetensors", "model.layers.41.mlp.gate_proj.weight": "model-00022-of-00051.safetensors", "model.layers.41.mlp.up_proj.weight": "model-00022-of-00051.safetensors", "model.layers.41.post_attention_layernorm.weight": "model-00022-of-00051.safetensors", "model.layers.41.self_attn.k_proj.weight": "model-00022-of-00051.safetensors", "model.layers.41.self_attn.o_proj.weight": "model-00022-of-00051.safetensors", "model.layers.41.self_attn.q_proj.weight": "model-00022-of-00051.safetensors", "model.layers.41.self_attn.v_proj.weight": "model-00022-of-00051.safetensors", "model.layers.42.input_layernorm.weight": "model-00022-of-00051.safetensors", "model.layers.42.mlp.down_proj.weight": "model-00022-of-00051.safetensors", "model.layers.42.mlp.gate_proj.weight": "model-00022-of-00051.safetensors", "model.layers.42.mlp.up_proj.weight": "model-00022-of-00051.safetensors", "model.layers.42.post_attention_layernorm.weight": "model-00022-of-00051.safetensors", "model.layers.42.self_attn.k_proj.weight": "model-00022-of-00051.safetensors", "model.layers.42.self_attn.o_proj.weight": "model-00023-of-00051.safetensors", "model.layers.42.self_attn.q_proj.weight": "model-00023-of-00051.safetensors", "model.layers.42.self_attn.v_proj.weight": "model-00023-of-00051.safetensors", "model.layers.43.input_layernorm.weight": "model-00023-of-00051.safetensors", "model.layers.43.mlp.down_proj.weight": "model-00023-of-00051.safetensors", "model.layers.43.mlp.gate_proj.weight": "model-00023-of-00051.safetensors", "model.layers.43.mlp.up_proj.weight": "model-00023-of-00051.safetensors", "model.layers.43.post_attention_layernorm.weight": "model-00023-of-00051.safetensors", "model.layers.43.self_attn.k_proj.weight": "model-00023-of-00051.safetensors", "model.layers.43.self_attn.o_proj.weight": "model-00023-of-00051.safetensors", "model.layers.43.self_attn.q_proj.weight": "model-00023-of-00051.safetensors", "model.layers.43.self_attn.v_proj.weight": "model-00023-of-00051.safetensors", "model.layers.44.input_layernorm.weight": "model-00023-of-00051.safetensors", "model.layers.44.mlp.down_proj.weight": "model-00023-of-00051.safetensors", "model.layers.44.mlp.gate_proj.weight": "model-00023-of-00051.safetensors", "model.layers.44.mlp.up_proj.weight": "model-00024-of-00051.safetensors", "model.layers.44.post_attention_layernorm.weight": "model-00024-of-00051.safetensors", "model.layers.44.self_attn.k_proj.weight": "model-00024-of-00051.safetensors", "model.layers.44.self_attn.o_proj.weight": "model-00024-of-00051.safetensors", "model.layers.44.self_attn.q_proj.weight": "model-00024-of-00051.safetensors", "model.layers.44.self_attn.v_proj.weight": "model-00024-of-00051.safetensors", "model.layers.45.input_layernorm.weight": "model-00024-of-00051.safetensors", "model.layers.45.mlp.down_proj.weight": "model-00024-of-00051.safetensors", "model.layers.45.mlp.gate_proj.weight": "model-00024-of-00051.safetensors", "model.layers.45.mlp.up_proj.weight": "model-00024-of-00051.safetensors", "model.layers.45.post_attention_layernorm.weight": "model-00024-of-00051.safetensors", "model.layers.45.self_attn.k_proj.weight": "model-00024-of-00051.safetensors", "model.layers.45.self_attn.o_proj.weight": "model-00024-of-00051.safetensors", "model.layers.45.self_attn.q_proj.weight": "model-00024-of-00051.safetensors", "model.layers.45.self_attn.v_proj.weight": "model-00024-of-00051.safetensors", "model.layers.46.input_layernorm.weight": "model-00024-of-00051.safetensors", "model.layers.46.mlp.down_proj.weight": "model-00024-of-00051.safetensors", "model.layers.46.mlp.gate_proj.weight": "model-00025-of-00051.safetensors", "model.layers.46.mlp.up_proj.weight": "model-00025-of-00051.safetensors", "model.layers.46.post_attention_layernorm.weight": "model-00025-of-00051.safetensors", "model.layers.46.self_attn.k_proj.weight": "model-00025-of-00051.safetensors", "model.layers.46.self_attn.o_proj.weight": "model-00025-of-00051.safetensors", "model.layers.46.self_attn.q_proj.weight": "model-00025-of-00051.safetensors", "model.layers.46.self_attn.v_proj.weight": "model-00025-of-00051.safetensors", "model.layers.47.input_layernorm.weight": "model-00025-of-00051.safetensors", "model.layers.47.mlp.down_proj.weight": "model-00025-of-00051.safetensors", "model.layers.47.mlp.gate_proj.weight": "model-00025-of-00051.safetensors", "model.layers.47.mlp.up_proj.weight": "model-00025-of-00051.safetensors", "model.layers.47.post_attention_layernorm.weight": "model-00025-of-00051.safetensors", "model.layers.47.self_attn.k_proj.weight": "model-00025-of-00051.safetensors", "model.layers.47.self_attn.o_proj.weight": "model-00025-of-00051.safetensors", "model.layers.47.self_attn.q_proj.weight": "model-00025-of-00051.safetensors", "model.layers.47.self_attn.v_proj.weight": "model-00025-of-00051.safetensors", "model.layers.48.input_layernorm.weight": "model-00025-of-00051.safetensors", "model.layers.48.mlp.down_proj.weight": "model-00026-of-00051.safetensors", "model.layers.48.mlp.gate_proj.weight": "model-00026-of-00051.safetensors", "model.layers.48.mlp.up_proj.weight": "model-00026-of-00051.safetensors", "model.layers.48.post_attention_layernorm.weight": "model-00026-of-00051.safetensors", "model.layers.48.self_attn.k_proj.weight": "model-00026-of-00051.safetensors", "model.layers.48.self_attn.o_proj.weight": "model-00026-of-00051.safetensors", "model.layers.48.self_attn.q_proj.weight": "model-00026-of-00051.safetensors", "model.layers.48.self_attn.v_proj.weight": "model-00026-of-00051.safetensors", "model.layers.49.input_layernorm.weight": "model-00026-of-00051.safetensors", "model.layers.49.mlp.down_proj.weight": "model-00026-of-00051.safetensors", "model.layers.49.mlp.gate_proj.weight": "model-00026-of-00051.safetensors", "model.layers.49.mlp.up_proj.weight": "model-00026-of-00051.safetensors", "model.layers.49.post_attention_layernorm.weight": "model-00026-of-00051.safetensors", "model.layers.49.self_attn.k_proj.weight": "model-00026-of-00051.safetensors", "model.layers.49.self_attn.o_proj.weight": "model-00027-of-00051.safetensors", "model.layers.49.self_attn.q_proj.weight": "model-00027-of-00051.safetensors", "model.layers.49.self_attn.v_proj.weight": "model-00027-of-00051.safetensors", "model.layers.5.input_layernorm.weight": "model-00027-of-00051.safetensors", "model.layers.5.mlp.down_proj.weight": "model-00027-of-00051.safetensors", "model.layers.5.mlp.gate_proj.weight": "model-00027-of-00051.safetensors", "model.layers.5.mlp.up_proj.weight": "model-00027-of-00051.safetensors", "model.layers.5.post_attention_layernorm.weight": "model-00027-of-00051.safetensors", "model.layers.5.self_attn.k_proj.weight": "model-00027-of-00051.safetensors", "model.layers.5.self_attn.o_proj.weight": "model-00027-of-00051.safetensors", "model.layers.5.self_attn.q_proj.weight": "model-00027-of-00051.safetensors", "model.layers.5.self_attn.v_proj.weight": "model-00027-of-00051.safetensors", "model.layers.50.input_layernorm.weight": "model-00027-of-00051.safetensors", "model.layers.50.mlp.down_proj.weight": "model-00027-of-00051.safetensors", "model.layers.50.mlp.gate_proj.weight": "model-00027-of-00051.safetensors", "model.layers.50.mlp.up_proj.weight": "model-00028-of-00051.safetensors", "model.layers.50.post_attention_layernorm.weight": "model-00028-of-00051.safetensors", "model.layers.50.self_attn.k_proj.weight": "model-00028-of-00051.safetensors", "model.layers.50.self_attn.o_proj.weight": "model-00028-of-00051.safetensors", "model.layers.50.self_attn.q_proj.weight": "model-00028-of-00051.safetensors", "model.layers.50.self_attn.v_proj.weight": "model-00028-of-00051.safetensors", "model.layers.51.input_layernorm.weight": "model-00028-of-00051.safetensors", "model.layers.51.mlp.down_proj.weight": "model-00028-of-00051.safetensors", "model.layers.51.mlp.gate_proj.weight": "model-00028-of-00051.safetensors", "model.layers.51.mlp.up_proj.weight": "model-00028-of-00051.safetensors", "model.layers.51.post_attention_layernorm.weight": "model-00028-of-00051.safetensors", "model.layers.51.self_attn.k_proj.weight": "model-00028-of-00051.safetensors", "model.layers.51.self_attn.o_proj.weight": "model-00028-of-00051.safetensors", "model.layers.51.self_attn.q_proj.weight": "model-00028-of-00051.safetensors", "model.layers.51.self_attn.v_proj.weight": "model-00028-of-00051.safetensors", "model.layers.52.input_layernorm.weight": "model-00028-of-00051.safetensors", "model.layers.52.mlp.down_proj.weight": "model-00028-of-00051.safetensors", "model.layers.52.mlp.gate_proj.weight": "model-00029-of-00051.safetensors", "model.layers.52.mlp.up_proj.weight": "model-00029-of-00051.safetensors", "model.layers.52.post_attention_layernorm.weight": "model-00029-of-00051.safetensors", "model.layers.52.self_attn.k_proj.weight": "model-00029-of-00051.safetensors", "model.layers.52.self_attn.o_proj.weight": "model-00029-of-00051.safetensors", "model.layers.52.self_attn.q_proj.weight": "model-00029-of-00051.safetensors", "model.layers.52.self_attn.v_proj.weight": "model-00029-of-00051.safetensors", "model.layers.53.input_layernorm.weight": "model-00029-of-00051.safetensors", "model.layers.53.mlp.down_proj.weight": "model-00029-of-00051.safetensors", "model.layers.53.mlp.gate_proj.weight": "model-00029-of-00051.safetensors", "model.layers.53.mlp.up_proj.weight": "model-00029-of-00051.safetensors", "model.layers.53.post_attention_layernorm.weight": "model-00029-of-00051.safetensors", "model.layers.53.self_attn.k_proj.weight": "model-00029-of-00051.safetensors", "model.layers.53.self_attn.o_proj.weight": "model-00029-of-00051.safetensors", "model.layers.53.self_attn.q_proj.weight": "model-00029-of-00051.safetensors", "model.layers.53.self_attn.v_proj.weight": "model-00029-of-00051.safetensors", "model.layers.54.input_layernorm.weight": "model-00029-of-00051.safetensors", "model.layers.54.mlp.down_proj.weight": "model-00030-of-00051.safetensors", "model.layers.54.mlp.gate_proj.weight": "model-00030-of-00051.safetensors", "model.layers.54.mlp.up_proj.weight": "model-00030-of-00051.safetensors", "model.layers.54.post_attention_layernorm.weight": "model-00030-of-00051.safetensors", "model.layers.54.self_attn.k_proj.weight": "model-00030-of-00051.safetensors", "model.layers.54.self_attn.o_proj.weight": "model-00030-of-00051.safetensors", "model.layers.54.self_attn.q_proj.weight": "model-00030-of-00051.safetensors", "model.layers.54.self_attn.v_proj.weight": "model-00030-of-00051.safetensors", "model.layers.55.input_layernorm.weight": "model-00030-of-00051.safetensors", "model.layers.55.mlp.down_proj.weight": "model-00030-of-00051.safetensors", "model.layers.55.mlp.gate_proj.weight": "model-00030-of-00051.safetensors", "model.layers.55.mlp.up_proj.weight": "model-00030-of-00051.safetensors", "model.layers.55.post_attention_layernorm.weight": "model-00030-of-00051.safetensors", "model.layers.55.self_attn.k_proj.weight": "model-00030-of-00051.safetensors", "model.layers.55.self_attn.o_proj.weight": "model-00031-of-00051.safetensors", "model.layers.55.self_attn.q_proj.weight": "model-00031-of-00051.safetensors", "model.layers.55.self_attn.v_proj.weight": "model-00031-of-00051.safetensors", "model.layers.56.input_layernorm.weight": "model-00031-of-00051.safetensors", "model.layers.56.mlp.down_proj.weight": "model-00031-of-00051.safetensors", "model.layers.56.mlp.gate_proj.weight": "model-00031-of-00051.safetensors", "model.layers.56.mlp.up_proj.weight": "model-00031-of-00051.safetensors", "model.layers.56.post_attention_layernorm.weight": "model-00031-of-00051.safetensors", "model.layers.56.self_attn.k_proj.weight": "model-00031-of-00051.safetensors", "model.layers.56.self_attn.o_proj.weight": "model-00031-of-00051.safetensors", "model.layers.56.self_attn.q_proj.weight": "model-00031-of-00051.safetensors", "model.layers.56.self_attn.v_proj.weight": "model-00031-of-00051.safetensors", "model.layers.57.input_layernorm.weight": "model-00031-of-00051.safetensors", "model.layers.57.mlp.down_proj.weight": "model-00031-of-00051.safetensors", "model.layers.57.mlp.gate_proj.weight": "model-00031-of-00051.safetensors", "model.layers.57.mlp.up_proj.weight": "model-00032-of-00051.safetensors", "model.layers.57.post_attention_layernorm.weight": "model-00032-of-00051.safetensors", "model.layers.57.self_attn.k_proj.weight": "model-00032-of-00051.safetensors", "model.layers.57.self_attn.o_proj.weight": "model-00032-of-00051.safetensors", "model.layers.57.self_attn.q_proj.weight": "model-00032-of-00051.safetensors", "model.layers.57.self_attn.v_proj.weight": "model-00032-of-00051.safetensors", "model.layers.58.input_layernorm.weight": "model-00032-of-00051.safetensors", "model.layers.58.mlp.down_proj.weight": "model-00032-of-00051.safetensors", "model.layers.58.mlp.gate_proj.weight": "model-00032-of-00051.safetensors", "model.layers.58.mlp.up_proj.weight": "model-00032-of-00051.safetensors", "model.layers.58.post_attention_layernorm.weight": "model-00032-of-00051.safetensors", "model.layers.58.self_attn.k_proj.weight": "model-00032-of-00051.safetensors", "model.layers.58.self_attn.o_proj.weight": "model-00032-of-00051.safetensors", "model.layers.58.self_attn.q_proj.weight": "model-00032-of-00051.safetensors", "model.layers.58.self_attn.v_proj.weight": "model-00032-of-00051.safetensors", "model.layers.59.input_layernorm.weight": "model-00032-of-00051.safetensors", "model.layers.59.mlp.down_proj.weight": "model-00032-of-00051.safetensors", "model.layers.59.mlp.gate_proj.weight": "model-00033-of-00051.safetensors", "model.layers.59.mlp.up_proj.weight": "model-00033-of-00051.safetensors", "model.layers.59.post_attention_layernorm.weight": "model-00033-of-00051.safetensors", "model.layers.59.self_attn.k_proj.weight": "model-00033-of-00051.safetensors", "model.layers.59.self_attn.o_proj.weight": "model-00033-of-00051.safetensors", "model.layers.59.self_attn.q_proj.weight": "model-00033-of-00051.safetensors", "model.layers.59.self_attn.v_proj.weight": "model-00033-of-00051.safetensors", "model.layers.6.input_layernorm.weight": "model-00033-of-00051.safetensors", "model.layers.6.mlp.down_proj.weight": "model-00033-of-00051.safetensors", "model.layers.6.mlp.gate_proj.weight": "model-00033-of-00051.safetensors", "model.layers.6.mlp.up_proj.weight": "model-00033-of-00051.safetensors", "model.layers.6.post_attention_layernorm.weight": "model-00033-of-00051.safetensors", "model.layers.6.self_attn.k_proj.weight": "model-00033-of-00051.safetensors", "model.layers.6.self_attn.o_proj.weight": "model-00033-of-00051.safetensors", "model.layers.6.self_attn.q_proj.weight": "model-00033-of-00051.safetensors", "model.layers.6.self_attn.v_proj.weight": "model-00033-of-00051.safetensors", "model.layers.60.input_layernorm.weight": "model-00033-of-00051.safetensors", "model.layers.60.mlp.down_proj.weight": "model-00034-of-00051.safetensors", "model.layers.60.mlp.gate_proj.weight": "model-00034-of-00051.safetensors", "model.layers.60.mlp.up_proj.weight": "model-00034-of-00051.safetensors", "model.layers.60.post_attention_layernorm.weight": "model-00034-of-00051.safetensors", "model.layers.60.self_attn.k_proj.weight": "model-00034-of-00051.safetensors", "model.layers.60.self_attn.o_proj.weight": "model-00034-of-00051.safetensors", "model.layers.60.self_attn.q_proj.weight": "model-00034-of-00051.safetensors", "model.layers.60.self_attn.v_proj.weight": "model-00034-of-00051.safetensors", "model.layers.61.input_layernorm.weight": "model-00034-of-00051.safetensors", "model.layers.61.mlp.down_proj.weight": "model-00034-of-00051.safetensors", "model.layers.61.mlp.gate_proj.weight": "model-00034-of-00051.safetensors", "model.layers.61.mlp.up_proj.weight": "model-00034-of-00051.safetensors", "model.layers.61.post_attention_layernorm.weight": "model-00034-of-00051.safetensors", "model.layers.61.self_attn.k_proj.weight": "model-00034-of-00051.safetensors", "model.layers.61.self_attn.o_proj.weight": "model-00035-of-00051.safetensors", "model.layers.61.self_attn.q_proj.weight": "model-00035-of-00051.safetensors", "model.layers.61.self_attn.v_proj.weight": "model-00035-of-00051.safetensors", "model.layers.62.input_layernorm.weight": "model-00035-of-00051.safetensors", "model.layers.62.mlp.down_proj.weight": "model-00035-of-00051.safetensors", "model.layers.62.mlp.gate_proj.weight": "model-00035-of-00051.safetensors", "model.layers.62.mlp.up_proj.weight": "model-00035-of-00051.safetensors", "model.layers.62.post_attention_layernorm.weight": "model-00035-of-00051.safetensors", "model.layers.62.self_attn.k_proj.weight": "model-00035-of-00051.safetensors", "model.layers.62.self_attn.o_proj.weight": "model-00035-of-00051.safetensors", "model.layers.62.self_attn.q_proj.weight": "model-00035-of-00051.safetensors", "model.layers.62.self_attn.v_proj.weight": "model-00035-of-00051.safetensors", "model.layers.63.input_layernorm.weight": "model-00035-of-00051.safetensors", "model.layers.63.mlp.down_proj.weight": "model-00035-of-00051.safetensors", "model.layers.63.mlp.gate_proj.weight": "model-00035-of-00051.safetensors", "model.layers.63.mlp.up_proj.weight": "model-00036-of-00051.safetensors", "model.layers.63.post_attention_layernorm.weight": "model-00036-of-00051.safetensors", "model.layers.63.self_attn.k_proj.weight": "model-00036-of-00051.safetensors", "model.layers.63.self_attn.o_proj.weight": "model-00036-of-00051.safetensors", "model.layers.63.self_attn.q_proj.weight": "model-00036-of-00051.safetensors", "model.layers.63.self_attn.v_proj.weight": "model-00036-of-00051.safetensors", "model.layers.64.input_layernorm.weight": "model-00036-of-00051.safetensors", "model.layers.64.mlp.down_proj.weight": "model-00036-of-00051.safetensors", "model.layers.64.mlp.gate_proj.weight": "model-00036-of-00051.safetensors", "model.layers.64.mlp.up_proj.weight": "model-00036-of-00051.safetensors", "model.layers.64.post_attention_layernorm.weight": "model-00036-of-00051.safetensors", "model.layers.64.self_attn.k_proj.weight": "model-00036-of-00051.safetensors", "model.layers.64.self_attn.o_proj.weight": "model-00036-of-00051.safetensors", "model.layers.64.self_attn.q_proj.weight": "model-00036-of-00051.safetensors", "model.layers.64.self_attn.v_proj.weight": "model-00036-of-00051.safetensors", "model.layers.65.input_layernorm.weight": "model-00036-of-00051.safetensors", "model.layers.65.mlp.down_proj.weight": "model-00036-of-00051.safetensors", "model.layers.65.mlp.gate_proj.weight": "model-00037-of-00051.safetensors", "model.layers.65.mlp.up_proj.weight": "model-00037-of-00051.safetensors", "model.layers.65.post_attention_layernorm.weight": "model-00037-of-00051.safetensors", "model.layers.65.self_attn.k_proj.weight": "model-00037-of-00051.safetensors", "model.layers.65.self_attn.o_proj.weight": "model-00037-of-00051.safetensors", "model.layers.65.self_attn.q_proj.weight": "model-00037-of-00051.safetensors", "model.layers.65.self_attn.v_proj.weight": "model-00037-of-00051.safetensors", "model.layers.66.input_layernorm.weight": "model-00037-of-00051.safetensors", "model.layers.66.mlp.down_proj.weight": "model-00037-of-00051.safetensors", "model.layers.66.mlp.gate_proj.weight": "model-00037-of-00051.safetensors", "model.layers.66.mlp.up_proj.weight": "model-00037-of-00051.safetensors", "model.layers.66.post_attention_layernorm.weight": "model-00037-of-00051.safetensors", "model.layers.66.self_attn.k_proj.weight": "model-00037-of-00051.safetensors", "model.layers.66.self_attn.o_proj.weight": "model-00037-of-00051.safetensors", "model.layers.66.self_attn.q_proj.weight": "model-00037-of-00051.safetensors", "model.layers.66.self_attn.v_proj.weight": "model-00037-of-00051.safetensors", "model.layers.67.input_layernorm.weight": "model-00037-of-00051.safetensors", "model.layers.67.mlp.down_proj.weight": "model-00038-of-00051.safetensors", "model.layers.67.mlp.gate_proj.weight": "model-00038-of-00051.safetensors", "model.layers.67.mlp.up_proj.weight": "model-00038-of-00051.safetensors", "model.layers.67.post_attention_layernorm.weight": "model-00038-of-00051.safetensors", "model.layers.67.self_attn.k_proj.weight": "model-00038-of-00051.safetensors", "model.layers.67.self_attn.o_proj.weight": "model-00038-of-00051.safetensors", "model.layers.67.self_attn.q_proj.weight": "model-00038-of-00051.safetensors", "model.layers.67.self_attn.v_proj.weight": "model-00038-of-00051.safetensors", "model.layers.68.input_layernorm.weight": "model-00038-of-00051.safetensors", "model.layers.68.mlp.down_proj.weight": "model-00038-of-00051.safetensors", "model.layers.68.mlp.gate_proj.weight": "model-00038-of-00051.safetensors", "model.layers.68.mlp.up_proj.weight": "model-00038-of-00051.safetensors", "model.layers.68.post_attention_layernorm.weight": "model-00038-of-00051.safetensors", "model.layers.68.self_attn.k_proj.weight": "model-00038-of-00051.safetensors", "model.layers.68.self_attn.o_proj.weight": "model-00039-of-00051.safetensors", "model.layers.68.self_attn.q_proj.weight": "model-00039-of-00051.safetensors", "model.layers.68.self_attn.v_proj.weight": "model-00039-of-00051.safetensors", "model.layers.69.input_layernorm.weight": "model-00039-of-00051.safetensors", "model.layers.69.mlp.down_proj.weight": "model-00039-of-00051.safetensors", "model.layers.69.mlp.gate_proj.weight": "model-00039-of-00051.safetensors", "model.layers.69.mlp.up_proj.weight": "model-00039-of-00051.safetensors", "model.layers.69.post_attention_layernorm.weight": "model-00039-of-00051.safetensors", "model.layers.69.self_attn.k_proj.weight": "model-00039-of-00051.safetensors", "model.layers.69.self_attn.o_proj.weight": "model-00039-of-00051.safetensors", "model.layers.69.self_attn.q_proj.weight": "model-00039-of-00051.safetensors", "model.layers.69.self_attn.v_proj.weight": "model-00039-of-00051.safetensors", "model.layers.7.input_layernorm.weight": "model-00039-of-00051.safetensors", "model.layers.7.mlp.down_proj.weight": "model-00039-of-00051.safetensors", "model.layers.7.mlp.gate_proj.weight": "model-00039-of-00051.safetensors", "model.layers.7.mlp.up_proj.weight": "model-00040-of-00051.safetensors", "model.layers.7.post_attention_layernorm.weight": "model-00040-of-00051.safetensors", "model.layers.7.self_attn.k_proj.weight": "model-00040-of-00051.safetensors", "model.layers.7.self_attn.o_proj.weight": "model-00040-of-00051.safetensors", "model.layers.7.self_attn.q_proj.weight": "model-00040-of-00051.safetensors", "model.layers.7.self_attn.v_proj.weight": "model-00040-of-00051.safetensors", "model.layers.70.input_layernorm.weight": "model-00040-of-00051.safetensors", "model.layers.70.mlp.down_proj.weight": "model-00040-of-00051.safetensors", "model.layers.70.mlp.gate_proj.weight": "model-00040-of-00051.safetensors", "model.layers.70.mlp.up_proj.weight": "model-00040-of-00051.safetensors", "model.layers.70.post_attention_layernorm.weight": "model-00040-of-00051.safetensors", "model.layers.70.self_attn.k_proj.weight": "model-00040-of-00051.safetensors", "model.layers.70.self_attn.o_proj.weight": "model-00040-of-00051.safetensors", "model.layers.70.self_attn.q_proj.weight": "model-00040-of-00051.safetensors", "model.layers.70.self_attn.v_proj.weight": "model-00040-of-00051.safetensors", "model.layers.71.input_layernorm.weight": "model-00040-of-00051.safetensors", "model.layers.71.mlp.down_proj.weight": "model-00040-of-00051.safetensors", "model.layers.71.mlp.gate_proj.weight": "model-00041-of-00051.safetensors", "model.layers.71.mlp.up_proj.weight": "model-00041-of-00051.safetensors", "model.layers.71.post_attention_layernorm.weight": "model-00041-of-00051.safetensors", "model.layers.71.self_attn.k_proj.weight": "model-00041-of-00051.safetensors", "model.layers.71.self_attn.o_proj.weight": "model-00041-of-00051.safetensors", "model.layers.71.self_attn.q_proj.weight": "model-00041-of-00051.safetensors", "model.layers.71.self_attn.v_proj.weight": "model-00041-of-00051.safetensors", "model.layers.72.input_layernorm.weight": "model-00041-of-00051.safetensors", "model.layers.72.mlp.down_proj.weight": "model-00041-of-00051.safetensors", "model.layers.72.mlp.gate_proj.weight": "model-00041-of-00051.safetensors", "model.layers.72.mlp.up_proj.weight": "model-00041-of-00051.safetensors", "model.layers.72.post_attention_layernorm.weight": "model-00041-of-00051.safetensors", "model.layers.72.self_attn.k_proj.weight": "model-00041-of-00051.safetensors", "model.layers.72.self_attn.o_proj.weight": "model-00041-of-00051.safetensors", "model.layers.72.self_attn.q_proj.weight": "model-00041-of-00051.safetensors", "model.layers.72.self_attn.v_proj.weight": "model-00041-of-00051.safetensors", "model.layers.73.input_layernorm.weight": "model-00041-of-00051.safetensors", "model.layers.73.mlp.down_proj.weight": "model-00042-of-00051.safetensors", "model.layers.73.mlp.gate_proj.weight": "model-00042-of-00051.safetensors", "model.layers.73.mlp.up_proj.weight": "model-00042-of-00051.safetensors", "model.layers.73.post_attention_layernorm.weight": "model-00042-of-00051.safetensors", "model.layers.73.self_attn.k_proj.weight": "model-00042-of-00051.safetensors", "model.layers.73.self_attn.o_proj.weight": "model-00042-of-00051.safetensors", "model.layers.73.self_attn.q_proj.weight": "model-00042-of-00051.safetensors", "model.layers.73.self_attn.v_proj.weight": "model-00042-of-00051.safetensors", "model.layers.74.input_layernorm.weight": "model-00042-of-00051.safetensors", "model.layers.74.mlp.down_proj.weight": "model-00042-of-00051.safetensors", "model.layers.74.mlp.gate_proj.weight": "model-00042-of-00051.safetensors", "model.layers.74.mlp.up_proj.weight": "model-00042-of-00051.safetensors", "model.layers.74.post_attention_layernorm.weight": "model-00042-of-00051.safetensors", "model.layers.74.self_attn.k_proj.weight": "model-00042-of-00051.safetensors", "model.layers.74.self_attn.o_proj.weight": "model-00043-of-00051.safetensors", "model.layers.74.self_attn.q_proj.weight": "model-00043-of-00051.safetensors", "model.layers.74.self_attn.v_proj.weight": "model-00043-of-00051.safetensors", "model.layers.75.input_layernorm.weight": "model-00043-of-00051.safetensors", "model.layers.75.mlp.down_proj.weight": "model-00043-of-00051.safetensors", "model.layers.75.mlp.gate_proj.weight": "model-00043-of-00051.safetensors", "model.layers.75.mlp.up_proj.weight": "model-00043-of-00051.safetensors", "model.layers.75.post_attention_layernorm.weight": "model-00043-of-00051.safetensors", "model.layers.75.self_attn.k_proj.weight": "model-00043-of-00051.safetensors", "model.layers.75.self_attn.o_proj.weight": "model-00043-of-00051.safetensors", "model.layers.75.self_attn.q_proj.weight": "model-00043-of-00051.safetensors", "model.layers.75.self_attn.v_proj.weight": "model-00043-of-00051.safetensors", "model.layers.76.input_layernorm.weight": "model-00043-of-00051.safetensors", "model.layers.76.mlp.down_proj.weight": "model-00043-of-00051.safetensors", "model.layers.76.mlp.gate_proj.weight": "model-00043-of-00051.safetensors", "model.layers.76.mlp.up_proj.weight": "model-00044-of-00051.safetensors", "model.layers.76.post_attention_layernorm.weight": "model-00044-of-00051.safetensors", "model.layers.76.self_attn.k_proj.weight": "model-00044-of-00051.safetensors", "model.layers.76.self_attn.o_proj.weight": "model-00044-of-00051.safetensors", "model.layers.76.self_attn.q_proj.weight": "model-00044-of-00051.safetensors", "model.layers.76.self_attn.v_proj.weight": "model-00044-of-00051.safetensors", "model.layers.77.input_layernorm.weight": "model-00044-of-00051.safetensors", "model.layers.77.mlp.down_proj.weight": "model-00044-of-00051.safetensors", "model.layers.77.mlp.gate_proj.weight": "model-00044-of-00051.safetensors", "model.layers.77.mlp.up_proj.weight": "model-00044-of-00051.safetensors", "model.layers.77.post_attention_layernorm.weight": "model-00044-of-00051.safetensors", "model.layers.77.self_attn.k_proj.weight": "model-00044-of-00051.safetensors", "model.layers.77.self_attn.o_proj.weight": "model-00044-of-00051.safetensors", "model.layers.77.self_attn.q_proj.weight": "model-00044-of-00051.safetensors", "model.layers.77.self_attn.v_proj.weight": "model-00044-of-00051.safetensors", "model.layers.78.input_layernorm.weight": "model-00044-of-00051.safetensors", "model.layers.78.mlp.down_proj.weight": "model-00044-of-00051.safetensors", "model.layers.78.mlp.gate_proj.weight": "model-00045-of-00051.safetensors", "model.layers.78.mlp.up_proj.weight": "model-00045-of-00051.safetensors", "model.layers.78.post_attention_layernorm.weight": "model-00045-of-00051.safetensors", "model.layers.78.self_attn.k_proj.weight": "model-00045-of-00051.safetensors", "model.layers.78.self_attn.o_proj.weight": "model-00045-of-00051.safetensors", "model.layers.78.self_attn.q_proj.weight": "model-00045-of-00051.safetensors", "model.layers.78.self_attn.v_proj.weight": "model-00045-of-00051.safetensors", "model.layers.79.input_layernorm.weight": "model-00045-of-00051.safetensors", "model.layers.79.mlp.down_proj.weight": "model-00045-of-00051.safetensors", "model.layers.79.mlp.gate_proj.weight": "model-00045-of-00051.safetensors", "model.layers.79.mlp.up_proj.weight": "model-00045-of-00051.safetensors", "model.layers.79.post_attention_layernorm.weight": "model-00045-of-00051.safetensors", "model.layers.79.self_attn.k_proj.weight": "model-00045-of-00051.safetensors", "model.layers.79.self_attn.o_proj.weight": "model-00045-of-00051.safetensors", "model.layers.79.self_attn.q_proj.weight": "model-00045-of-00051.safetensors", "model.layers.79.self_attn.v_proj.weight": "model-00045-of-00051.safetensors", "model.layers.8.input_layernorm.weight": "model-00045-of-00051.safetensors", "model.layers.8.mlp.down_proj.weight": "model-00046-of-00051.safetensors", "model.layers.8.mlp.gate_proj.weight": "model-00046-of-00051.safetensors", "model.layers.8.mlp.up_proj.weight": "model-00046-of-00051.safetensors", "model.layers.8.post_attention_layernorm.weight": "model-00046-of-00051.safetensors", "model.layers.8.self_attn.k_proj.weight": "model-00046-of-00051.safetensors", "model.layers.8.self_attn.o_proj.weight": "model-00046-of-00051.safetensors", "model.layers.8.self_attn.q_proj.weight": "model-00046-of-00051.safetensors", "model.layers.8.self_attn.v_proj.weight": "model-00046-of-00051.safetensors", "model.layers.80.input_layernorm.weight": "model-00046-of-00051.safetensors", "model.layers.80.mlp.down_proj.weight": "model-00046-of-00051.safetensors", "model.layers.80.mlp.gate_proj.weight": "model-00046-of-00051.safetensors", "model.layers.80.mlp.up_proj.weight": "model-00046-of-00051.safetensors", "model.layers.80.post_attention_layernorm.weight": "model-00046-of-00051.safetensors", "model.layers.80.self_attn.k_proj.weight": "model-00046-of-00051.safetensors", "model.layers.80.self_attn.o_proj.weight": "model-00047-of-00051.safetensors", "model.layers.80.self_attn.q_proj.weight": "model-00047-of-00051.safetensors", "model.layers.80.self_attn.v_proj.weight": "model-00047-of-00051.safetensors", "model.layers.81.input_layernorm.weight": "model-00047-of-00051.safetensors", "model.layers.81.mlp.down_proj.weight": "model-00047-of-00051.safetensors", "model.layers.81.mlp.gate_proj.weight": "model-00047-of-00051.safetensors", "model.layers.81.mlp.up_proj.weight": "model-00047-of-00051.safetensors", "model.layers.81.post_attention_layernorm.weight": "model-00047-of-00051.safetensors", "model.layers.81.self_attn.k_proj.weight": "model-00047-of-00051.safetensors", "model.layers.81.self_attn.o_proj.weight": "model-00047-of-00051.safetensors", "model.layers.81.self_attn.q_proj.weight": "model-00047-of-00051.safetensors", "model.layers.81.self_attn.v_proj.weight": "model-00047-of-00051.safetensors", "model.layers.82.input_layernorm.weight": "model-00047-of-00051.safetensors", "model.layers.82.mlp.down_proj.weight": "model-00047-of-00051.safetensors", "model.layers.82.mlp.gate_proj.weight": "model-00047-of-00051.safetensors", "model.layers.82.mlp.up_proj.weight": "model-00048-of-00051.safetensors", "model.layers.82.post_attention_layernorm.weight": "model-00048-of-00051.safetensors", "model.layers.82.self_attn.k_proj.weight": "model-00048-of-00051.safetensors", "model.layers.82.self_attn.o_proj.weight": "model-00048-of-00051.safetensors", "model.layers.82.self_attn.q_proj.weight": "model-00048-of-00051.safetensors", "model.layers.82.self_attn.v_proj.weight": "model-00048-of-00051.safetensors", "model.layers.83.input_layernorm.weight": "model-00048-of-00051.safetensors", "model.layers.83.mlp.down_proj.weight": "model-00048-of-00051.safetensors", "model.layers.83.mlp.gate_proj.weight": "model-00048-of-00051.safetensors", "model.layers.83.mlp.up_proj.weight": "model-00048-of-00051.safetensors", "model.layers.83.post_attention_layernorm.weight": "model-00048-of-00051.safetensors", "model.layers.83.self_attn.k_proj.weight": "model-00048-of-00051.safetensors", "model.layers.83.self_attn.o_proj.weight": "model-00048-of-00051.safetensors", "model.layers.83.self_attn.q_proj.weight": "model-00048-of-00051.safetensors", "model.layers.83.self_attn.v_proj.weight": "model-00048-of-00051.safetensors", "model.layers.84.input_layernorm.weight": "model-00048-of-00051.safetensors", "model.layers.84.mlp.down_proj.weight": "model-00048-of-00051.safetensors", "model.layers.84.mlp.gate_proj.weight": "model-00049-of-00051.safetensors", "model.layers.84.mlp.up_proj.weight": "model-00049-of-00051.safetensors", "model.layers.84.post_attention_layernorm.weight": "model-00049-of-00051.safetensors", "model.layers.84.self_attn.k_proj.weight": "model-00049-of-00051.safetensors", "model.layers.84.self_attn.o_proj.weight": "model-00049-of-00051.safetensors", "model.layers.84.self_attn.q_proj.weight": "model-00049-of-00051.safetensors", "model.layers.84.self_attn.v_proj.weight": "model-00049-of-00051.safetensors", "model.layers.85.input_layernorm.weight": "model-00049-of-00051.safetensors", "model.layers.85.mlp.down_proj.weight": "model-00049-of-00051.safetensors", "model.layers.85.mlp.gate_proj.weight": "model-00049-of-00051.safetensors", "model.layers.85.mlp.up_proj.weight": "model-00049-of-00051.safetensors", "model.layers.85.post_attention_layernorm.weight": "model-00049-of-00051.safetensors", "model.layers.85.self_attn.k_proj.weight": "model-00049-of-00051.safetensors", "model.layers.85.self_attn.o_proj.weight": "model-00049-of-00051.safetensors", "model.layers.85.self_attn.q_proj.weight": "model-00049-of-00051.safetensors", "model.layers.85.self_attn.v_proj.weight": "model-00049-of-00051.safetensors", "model.layers.86.input_layernorm.weight": "model-00049-of-00051.safetensors", "model.layers.86.mlp.down_proj.weight": "model-00050-of-00051.safetensors", "model.layers.86.mlp.gate_proj.weight": "model-00050-of-00051.safetensors", "model.layers.86.mlp.up_proj.weight": "model-00050-of-00051.safetensors", "model.layers.86.post_attention_layernorm.weight": "model-00050-of-00051.safetensors", "model.layers.86.self_attn.k_proj.weight": "model-00050-of-00051.safetensors", "model.layers.86.self_attn.o_proj.weight": "model-00050-of-00051.safetensors", "model.layers.86.self_attn.q_proj.weight": "model-00050-of-00051.safetensors", "model.layers.86.self_attn.v_proj.weight": "model-00050-of-00051.safetensors", "model.layers.87.input_layernorm.weight": "model-00050-of-00051.safetensors", "model.layers.87.mlp.down_proj.weight": "model-00050-of-00051.safetensors", "model.layers.87.mlp.gate_proj.weight": "model-00050-of-00051.safetensors", "model.layers.87.mlp.up_proj.weight": "model-00050-of-00051.safetensors", "model.layers.87.post_attention_layernorm.weight": "model-00050-of-00051.safetensors", "model.layers.87.self_attn.k_proj.weight": "model-00050-of-00051.safetensors", "model.layers.87.self_attn.o_proj.weight": "model-00051-of-00051.safetensors", "model.layers.87.self_attn.q_proj.weight": "model-00051-of-00051.safetensors", "model.layers.87.self_attn.v_proj.weight": "model-00051-of-00051.safetensors", "model.layers.9.input_layernorm.weight": "model-00051-of-00051.safetensors", "model.layers.9.mlp.down_proj.weight": "model-00051-of-00051.safetensors", "model.layers.9.mlp.gate_proj.weight": "model-00051-of-00051.safetensors", "model.layers.9.mlp.up_proj.weight": "model-00051-of-00051.safetensors", "model.layers.9.post_attention_layernorm.weight": "model-00051-of-00051.safetensors", "model.layers.9.self_attn.k_proj.weight": "model-00051-of-00051.safetensors", "model.layers.9.self_attn.o_proj.weight": "model-00051-of-00051.safetensors", "model.layers.9.self_attn.q_proj.weight": "model-00051-of-00051.safetensors", "model.layers.9.self_attn.v_proj.weight": "model-00051-of-00051.safetensors", "model.norm.weight": "model-00051-of-00051.safetensors"}}
|
special_tokens_map.json
ADDED
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"bos_token": {
|
3 |
+
"content": "<s>",
|
4 |
+
"lstrip": false,
|
5 |
+
"normalized": false,
|
6 |
+
"rstrip": false,
|
7 |
+
"single_word": false
|
8 |
+
},
|
9 |
+
"eos_token": {
|
10 |
+
"content": "</s>",
|
11 |
+
"lstrip": false,
|
12 |
+
"normalized": false,
|
13 |
+
"rstrip": false,
|
14 |
+
"single_word": false
|
15 |
+
},
|
16 |
+
"unk_token": {
|
17 |
+
"content": "<unk>",
|
18 |
+
"lstrip": false,
|
19 |
+
"normalized": false,
|
20 |
+
"rstrip": false,
|
21 |
+
"single_word": false
|
22 |
+
}
|
23 |
+
}
|
tokenizer.json
ADDED
The diff for this file is too large to render.
See raw diff
|
|
tokenizer.model
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:59f95e28944c062244741268596badc900df86c7f5ded05088d2da22a7379e06
|
3 |
+
size 587583
|
tokenizer_config.json
ADDED
The diff for this file is too large to render.
See raw diff
|
|