maeeeeee commited on
Commit
b084e6a
·
verified ·
1 Parent(s): 1a2756b

Delete https:

Browse files
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/.gitattributes DELETED
@@ -1,35 +0,0 @@
1
- *.7z filter=lfs diff=lfs merge=lfs -text
2
- *.arrow filter=lfs diff=lfs merge=lfs -text
3
- *.bin filter=lfs diff=lfs merge=lfs -text
4
- *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
- *.ftz filter=lfs diff=lfs merge=lfs -text
7
- *.gz filter=lfs diff=lfs merge=lfs -text
8
- *.h5 filter=lfs diff=lfs merge=lfs -text
9
- *.joblib filter=lfs diff=lfs merge=lfs -text
10
- *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
- *.model filter=lfs diff=lfs merge=lfs -text
13
- *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
- *.onnx filter=lfs diff=lfs merge=lfs -text
17
- *.ot filter=lfs diff=lfs merge=lfs -text
18
- *.parquet filter=lfs diff=lfs merge=lfs -text
19
- *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
- *.pt filter=lfs diff=lfs merge=lfs -text
23
- *.pth filter=lfs diff=lfs merge=lfs -text
24
- *.rar filter=lfs diff=lfs merge=lfs -text
25
- *.safetensors filter=lfs diff=lfs merge=lfs -text
26
- saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
- *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
- *.tflite filter=lfs diff=lfs merge=lfs -text
30
- *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
- *.xz filter=lfs diff=lfs merge=lfs -text
33
- *.zip filter=lfs diff=lfs merge=lfs -text
34
- *.zst filter=lfs diff=lfs merge=lfs -text
35
- *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/README.md DELETED
@@ -1,67 +0,0 @@
1
- kudos to that one dude on /lmg/ for shouting out this model because I dig it. this is my first time uploading something so if something on HF is broken I simply do not care
2
-
3
- ---
4
- base_model:
5
- - mistralai/Mixtral-8x7B-v0.1
6
- - mistralai/Mixtral-8x7B-Instruct-v0.1
7
- - jondurbin/bagel-dpo-8x7b-v0.2
8
- - cognitivecomputations/dolphin-2.7-mixtral-8x7b
9
- - NeverSleep/Noromaid-v0.4-Mixtral-Instruct-8x7b-Zloss
10
- - ycros/BagelMIsteryTour-v2-8x7B
11
- - smelborp/MixtralOrochi8x7B
12
- library_name: transformers
13
- tags:
14
- - mergekit
15
- - merge
16
-
17
- ---
18
- # maid-yuzu-v8-alter
19
-
20
- This is a merge of pre-trained language models created using [mergekit](https://github.com/cg123/mergekit).
21
-
22
- v7's approach worked better than I thought, so I tried something even weirder as a test. I don't think a proper model will come out, but I'm curious about the results.
23
-
24
- ## Merge Details
25
- ### Merge Method
26
-
27
- This model was merged using the SLERP merge method.
28
-
29
- This models were merged using the SLERP method in the following order:
30
-
31
- maid-yuzu-v8-base: mistralai/Mixtral-8x7B-v0.1 + mistralai/Mixtral-8x7B-Instruct-v0.1 = 0.5
32
- maid-yuzu-v8-step1: above + jondurbin/bagel-dpo-8x7b-v0.2 = 0.25
33
- maid-yuzu-v8-step2: above + cognitivecomputations/dolphin-2.7-mixtral-8x7b = 0.25
34
- maid-yuzu-v8-step3: above + NeverSleep/Noromaid-v0.4-Mixtral-Instruct-8x7b-Zloss = 0.25
35
- maid-yuzu-v8-step4-alter: above + ycros/BagelMIsteryTour-v2-8x7B = 0.5
36
- maid-yuzu-v8-alter: above + smelborp/MixtralOrochi8x7B = 0.5
37
-
38
- ### Models Merged
39
-
40
- The following models were included in the merge:
41
- * [smelborp/MixtralOrochi8x7B](https://huggingface.co/smelborp/MixtralOrochi8x7B)
42
- * ../maid-yuzu-v8-step4-alter
43
-
44
- ### Configuration
45
-
46
- The following YAML configuration was used to produce this model:
47
-
48
- ```yaml
49
- base_model:
50
- model:
51
- path: ../maid-yuzu-v8-step4-alter
52
- dtype: bfloat16
53
- merge_method: slerp
54
- parameters:
55
- t:
56
- - value: 0.5
57
- slices:
58
- - sources:
59
- - layer_range: [0, 32]
60
- model:
61
- model:
62
- path: ../maid-yuzu-v8-step4-alter
63
- - layer_range: [0, 32]
64
- model:
65
- model:
66
- path: smelborp/MixtralOrochi8x7B
67
- ```
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/config.json DELETED
@@ -1,30 +0,0 @@
1
- {
2
- "_name_or_path": "../maid-yuzu-v8-step4-alter",
3
- "architectures": [
4
- "MixtralForCausalLM"
5
- ],
6
- "attention_dropout": 0.0,
7
- "bos_token_id": 1,
8
- "eos_token_id": 2,
9
- "hidden_act": "silu",
10
- "hidden_size": 4096,
11
- "initializer_range": 0.02,
12
- "intermediate_size": 14336,
13
- "max_position_embeddings": 32768,
14
- "model_type": "mixtral",
15
- "num_attention_heads": 32,
16
- "num_experts_per_tok": 2,
17
- "num_hidden_layers": 32,
18
- "num_key_value_heads": 8,
19
- "num_local_experts": 8,
20
- "output_router_logits": false,
21
- "rms_norm_eps": 1e-05,
22
- "rope_theta": 1000000.0,
23
- "router_aux_loss_coef": 0.02,
24
- "sliding_window": null,
25
- "tie_word_embeddings": false,
26
- "torch_dtype": "bfloat16",
27
- "transformers_version": "4.37.2",
28
- "use_cache": true,
29
- "vocab_size": 32000
30
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/mergekit_config.yml DELETED
@@ -1,18 +0,0 @@
1
- base_model:
2
- model:
3
- path: ../maid-yuzu-v8-step4-alter
4
- dtype: bfloat16
5
- merge_method: slerp
6
- parameters:
7
- t:
8
- - value: 0.5
9
- slices:
10
- - sources:
11
- - layer_range: [0, 32]
12
- model:
13
- model:
14
- path: ../maid-yuzu-v8-step4-alter
15
- - layer_range: [0, 32]
16
- model:
17
- model:
18
- path: smelborp/MixtralOrochi8x7B
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/model.safetensors.index.json DELETED
@@ -1 +0,0 @@
1
- {"metadata": {"mergekit_version": "0.0.4"}, "weight_map": {"model.layers.3.block_sparse_moe.gate.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.gate.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.4.w2.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.4.w1.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.3.w3.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.3.w2.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.3.w1.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.2.w3.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.2.w2.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.2.w1.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.1.w3.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.1.w2.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.1.w1.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.0.w3.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.0.w2.weight": "model-00001-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.0.w1.weight": "model-00001-of-00048.safetensors", "model.layers.1.self_attn.o_proj.weight": "model-00001-of-00048.safetensors", "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00048.safetensors", "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00048.safetensors", "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00048.safetensors", "model.layers.0.block_sparse_moe.gate.weight": "model-00001-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.7.w3.weight": "model-00001-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.7.w2.weight": "model-00001-of-00048.safetensors", "model.layers.5.block_sparse_moe.gate.weight": "model-00001-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.7.w1.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.6.w3.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.6.w2.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.6.w1.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.5.w3.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.5.w2.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.5.w1.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.4.w3.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.4.w2.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.4.w1.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.3.w3.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.3.w2.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.3.w1.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.2.w3.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.2.w2.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.2.w1.weight": "model-00002-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.1.w3.weight": "model-00002-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.2.w1.weight": "model-00003-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.1.w3.weight": "model-00003-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.1.w2.weight": "model-00003-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.1.w1.weight": "model-00003-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.0.w3.weight": "model-00003-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.0.w2.weight": "model-00003-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.0.w1.weight": "model-00003-of-00048.safetensors", "model.layers.3.self_attn.o_proj.weight": "model-00003-of-00048.safetensors", "model.layers.3.self_attn.v_proj.weight": "model-00003-of-00048.safetensors", "model.layers.3.self_attn.k_proj.weight": "model-00003-of-00048.safetensors", "model.layers.3.self_attn.q_proj.weight": "model-00003-of-00048.safetensors", "model.layers.2.block_sparse_moe.gate.weight": "model-00003-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.7.w3.weight": "model-00003-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.1.w2.weight": "model-00003-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.1.w1.weight": "model-00003-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.0.w3.weight": "model-00003-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.0.w2.weight": "model-00003-of-00048.safetensors", "model.layers.0.block_sparse_moe.experts.0.w1.weight": "model-00003-of-00048.safetensors", "model.layers.0.self_attn.o_proj.weight": "model-00003-of-00048.safetensors", "model.layers.0.self_attn.v_proj.weight": "model-00003-of-00048.safetensors", "model.layers.0.self_attn.k_proj.weight": "model-00003-of-00048.safetensors", "model.layers.0.self_attn.q_proj.weight": "model-00003-of-00048.safetensors", "model.layers.0.post_attention_layernorm.weight": "model-00003-of-00048.safetensors", "model.layers.0.input_layernorm.weight": "model-00003-of-00048.safetensors", "model.embed_tokens.weight": "model-00003-of-00048.safetensors", "model.layers.6.block_sparse_moe.gate.weight": "model-00003-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.7.w2.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.7.w1.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.6.w3.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.6.w2.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.6.w1.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.5.w3.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.5.w2.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.5.w1.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.4.w3.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.4.w2.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.4.w1.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.3.w3.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.3.w2.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.3.w1.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.2.w3.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.2.w2.weight": "model-00004-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.2.w1.weight": "model-00004-of-00048.safetensors", "model.layers.5.self_attn.o_proj.weight": "model-00005-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.1.w3.weight": "model-00005-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.1.w2.weight": "model-00005-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.1.w1.weight": "model-00005-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.0.w3.weight": "model-00005-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.0.w2.weight": "model-00005-of-00048.safetensors", "model.layers.2.block_sparse_moe.experts.0.w1.weight": "model-00005-of-00048.safetensors", "model.layers.2.self_attn.o_proj.weight": "model-00005-of-00048.safetensors", "model.layers.2.self_attn.v_proj.weight": "model-00005-of-00048.safetensors", "model.layers.2.self_attn.k_proj.weight": "model-00005-of-00048.safetensors", "model.layers.2.self_attn.q_proj.weight": "model-00005-of-00048.safetensors", "model.layers.2.post_attention_layernorm.weight": "model-00005-of-00048.safetensors", "model.layers.2.input_layernorm.weight": "model-00005-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.7.w3.weight": "model-00005-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.7.w2.weight": "model-00005-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.7.w1.weight": "model-00005-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.6.w3.weight": "model-00005-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.6.w2.weight": "model-00005-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.6.w1.weight": "model-00005-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.5.w3.weight": "model-00005-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.5.w2.weight": "model-00005-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.5.w1.weight": "model-00005-of-00048.safetensors", "model.layers.1.block_sparse_moe.experts.4.w3.weight": "model-00005-of-00048.safetensors", "model.layers.1.post_attention_layernorm.weight": "model-00005-of-00048.safetensors", "model.layers.1.input_layernorm.weight": "model-00005-of-00048.safetensors", "model.layers.8.block_sparse_moe.gate.weight": "model-00005-of-00048.safetensors", "model.layers.5.self_attn.v_proj.weight": "model-00006-of-00048.safetensors", "model.layers.5.self_attn.k_proj.weight": "model-00006-of-00048.safetensors", "model.layers.5.self_attn.q_proj.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.gate.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.7.w3.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.7.w2.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.7.w1.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.6.w3.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.6.w2.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.6.w1.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.5.w3.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.5.w2.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.5.w1.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.4.w3.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.4.w2.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.4.w1.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.3.w3.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.3.w2.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.3.w1.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.2.w3.weight": "model-00006-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.2.w2.weight": "model-00007-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.2.w1.weight": "model-00007-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.1.w3.weight": "model-00007-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.1.w2.weight": "model-00007-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.1.w1.weight": "model-00007-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.0.w3.weight": "model-00007-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.0.w2.weight": "model-00007-of-00048.safetensors", "model.layers.4.block_sparse_moe.experts.0.w1.weight": "model-00007-of-00048.safetensors", "model.layers.4.self_attn.o_proj.weight": "model-00007-of-00048.safetensors", "model.layers.4.self_attn.v_proj.weight": "model-00007-of-00048.safetensors", "model.layers.4.self_attn.k_proj.weight": "model-00007-of-00048.safetensors", "model.layers.4.self_attn.q_proj.weight": "model-00007-of-00048.safetensors", "model.layers.4.post_attention_layernorm.weight": "model-00007-of-00048.safetensors", "model.layers.4.input_layernorm.weight": "model-00007-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.7.w3.weight": "model-00007-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.7.w2.weight": "model-00007-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.7.w1.weight": "model-00007-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.6.w3.weight": "model-00007-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.6.w2.weight": "model-00007-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.6.w1.weight": "model-00007-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.5.w3.weight": "model-00007-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.5.w2.weight": "model-00007-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.5.w2.weight": "model-00008-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.5.w1.weight": "model-00008-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.4.w3.weight": "model-00008-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.4.w2.weight": "model-00008-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.4.w1.weight": "model-00008-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.3.w3.weight": "model-00008-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.3.w2.weight": "model-00008-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.3.w1.weight": "model-00008-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.5.w1.weight": "model-00008-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.4.w3.weight": "model-00008-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.4.w2.weight": "model-00008-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.4.w1.weight": "model-00008-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.3.w3.weight": "model-00008-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.3.w2.weight": "model-00008-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.3.w1.weight": "model-00008-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.2.w3.weight": "model-00008-of-00048.safetensors", "model.layers.3.block_sparse_moe.experts.2.w2.weight": "model-00008-of-00048.safetensors", "model.layers.3.post_attention_layernorm.weight": "model-00008-of-00048.safetensors", "model.layers.3.input_layernorm.weight": "model-00008-of-00048.safetensors", "model.layers.10.block_sparse_moe.gate.weight": "model-00008-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.2.w3.weight": "model-00009-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.2.w2.weight": "model-00009-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.2.w1.weight": "model-00009-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.1.w3.weight": "model-00009-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.1.w2.weight": "model-00009-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.1.w1.weight": "model-00009-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.0.w3.weight": "model-00009-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.0.w2.weight": "model-00009-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.0.w1.weight": "model-00009-of-00048.safetensors", "model.layers.6.self_attn.o_proj.weight": "model-00009-of-00048.safetensors", "model.layers.6.self_attn.v_proj.weight": "model-00009-of-00048.safetensors", "model.layers.6.self_attn.k_proj.weight": "model-00009-of-00048.safetensors", "model.layers.6.self_attn.q_proj.weight": "model-00009-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.7.w3.weight": "model-00009-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.7.w2.weight": "model-00009-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.7.w1.weight": "model-00009-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.6.w3.weight": "model-00009-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.6.w2.weight": "model-00009-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.6.w1.weight": "model-00009-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.5.w3.weight": "model-00009-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.5.w2.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.5.w1.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.4.w3.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.4.w2.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.4.w1.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.3.w3.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.3.w2.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.3.w1.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.2.w3.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.2.w2.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.2.w1.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.1.w3.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.1.w2.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.1.w1.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.0.w3.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.0.w2.weight": "model-00010-of-00048.safetensors", "model.layers.5.block_sparse_moe.experts.0.w1.weight": "model-00010-of-00048.safetensors", "model.layers.5.post_attention_layernorm.weight": "model-00010-of-00048.safetensors", "model.layers.5.input_layernorm.weight": "model-00010-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.3.w1.weight": "model-00011-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.2.w3.weight": "model-00011-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.2.w2.weight": "model-00011-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.2.w1.weight": "model-00011-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.1.w3.weight": "model-00011-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.1.w2.weight": "model-00011-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.1.w1.weight": "model-00011-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.0.w3.weight": "model-00011-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.0.w2.weight": "model-00011-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.0.w1.weight": "model-00011-of-00048.safetensors", "model.layers.8.self_attn.o_proj.weight": "model-00011-of-00048.safetensors", "model.layers.8.self_attn.v_proj.weight": "model-00011-of-00048.safetensors", "model.layers.8.self_attn.k_proj.weight": "model-00011-of-00048.safetensors", "model.layers.8.self_attn.q_proj.weight": "model-00011-of-00048.safetensors", "model.layers.7.block_sparse_moe.gate.weight": "model-00011-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.7.w3.weight": "model-00011-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.7.w2.weight": "model-00011-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.7.w1.weight": "model-00011-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.6.w3.weight": "model-00011-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.6.w2.weight": "model-00011-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.6.w1.weight": "model-00011-of-00048.safetensors", "model.layers.11.block_sparse_moe.gate.weight": "model-00011-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.5.w3.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.5.w2.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.5.w1.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.4.w3.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.4.w2.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.4.w1.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.3.w3.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.3.w2.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.3.w1.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.2.w3.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.2.w2.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.2.w1.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.1.w3.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.1.w2.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.1.w1.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.0.w3.weight": "model-00012-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.0.w2.weight": "model-00012-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.0.w3.weight": "model-00013-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.0.w2.weight": "model-00013-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.0.w1.weight": "model-00013-of-00048.safetensors", "model.layers.10.self_attn.o_proj.weight": "model-00013-of-00048.safetensors", "model.layers.10.self_attn.v_proj.weight": "model-00013-of-00048.safetensors", "model.layers.10.self_attn.k_proj.weight": "model-00013-of-00048.safetensors", "model.layers.10.self_attn.q_proj.weight": "model-00013-of-00048.safetensors", "model.layers.9.block_sparse_moe.gate.weight": "model-00013-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.7.w3.weight": "model-00013-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.7.w2.weight": "model-00013-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.7.w1.weight": "model-00013-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.6.w3.weight": "model-00013-of-00048.safetensors", "model.layers.7.block_sparse_moe.experts.0.w1.weight": "model-00013-of-00048.safetensors", "model.layers.7.self_attn.o_proj.weight": "model-00013-of-00048.safetensors", "model.layers.7.self_attn.v_proj.weight": "model-00013-of-00048.safetensors", "model.layers.7.self_attn.k_proj.weight": "model-00013-of-00048.safetensors", "model.layers.7.self_attn.q_proj.weight": "model-00013-of-00048.safetensors", "model.layers.7.post_attention_layernorm.weight": "model-00013-of-00048.safetensors", "model.layers.7.input_layernorm.weight": "model-00013-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.7.w3.weight": "model-00013-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.7.w2.weight": "model-00013-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.7.w1.weight": "model-00013-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.6.w3.weight": "model-00013-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.6.w2.weight": "model-00013-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.6.w1.weight": "model-00013-of-00048.safetensors", "model.layers.6.block_sparse_moe.experts.5.w3.weight": "model-00013-of-00048.safetensors", "model.layers.6.post_attention_layernorm.weight": "model-00013-of-00048.safetensors", "model.layers.6.input_layernorm.weight": "model-00013-of-00048.safetensors", "model.layers.13.block_sparse_moe.gate.weight": "model-00013-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.6.w2.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.6.w1.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.5.w3.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.5.w2.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.5.w1.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.4.w3.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.4.w2.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.4.w1.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.3.w3.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.3.w2.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.3.w1.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.2.w3.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.2.w2.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.2.w1.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.1.w3.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.1.w2.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.1.w1.weight": "model-00014-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.0.w3.weight": "model-00015-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.0.w2.weight": "model-00015-of-00048.safetensors", "model.layers.9.block_sparse_moe.experts.0.w1.weight": "model-00015-of-00048.safetensors", "model.layers.9.self_attn.o_proj.weight": "model-00015-of-00048.safetensors", "model.layers.9.self_attn.v_proj.weight": "model-00015-of-00048.safetensors", "model.layers.9.self_attn.k_proj.weight": "model-00015-of-00048.safetensors", "model.layers.9.self_attn.q_proj.weight": "model-00015-of-00048.safetensors", "model.layers.9.post_attention_layernorm.weight": "model-00015-of-00048.safetensors", "model.layers.9.input_layernorm.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.7.w3.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.7.w2.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.7.w1.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.6.w3.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.6.w2.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.6.w1.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.5.w3.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.5.w2.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.5.w1.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.4.w3.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.4.w2.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.4.w1.weight": "model-00015-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.3.w3.weight": "model-00015-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.6.w2.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.6.w1.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.5.w3.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.5.w2.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.5.w1.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.4.w3.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.4.w2.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.4.w1.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.3.w3.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.3.w2.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.3.w1.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.2.w3.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.2.w2.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.2.w1.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.1.w3.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.1.w2.weight": "model-00016-of-00048.safetensors", "model.layers.8.block_sparse_moe.experts.3.w2.weight": "model-00016-of-00048.safetensors", "model.layers.8.post_attention_layernorm.weight": "model-00016-of-00048.safetensors", "model.layers.8.input_layernorm.weight": "model-00016-of-00048.safetensors", "model.layers.15.block_sparse_moe.gate.weight": "model-00016-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.1.w1.weight": "model-00017-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.0.w3.weight": "model-00017-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.0.w2.weight": "model-00017-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.0.w1.weight": "model-00017-of-00048.safetensors", "model.layers.11.self_attn.o_proj.weight": "model-00017-of-00048.safetensors", "model.layers.11.self_attn.v_proj.weight": "model-00017-of-00048.safetensors", "model.layers.11.self_attn.k_proj.weight": "model-00017-of-00048.safetensors", "model.layers.11.self_attn.q_proj.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.7.w3.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.7.w2.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.7.w1.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.6.w3.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.6.w2.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.6.w1.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.5.w3.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.5.w2.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.5.w1.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.4.w3.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.4.w2.weight": "model-00017-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.4.w1.weight": "model-00017-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.4.w1.weight": "model-00018-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.3.w3.weight": "model-00018-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.3.w2.weight": "model-00018-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.3.w1.weight": "model-00018-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.2.w3.weight": "model-00018-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.2.w2.weight": "model-00018-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.2.w1.weight": "model-00018-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.1.w3.weight": "model-00018-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.3.w3.weight": "model-00018-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.3.w2.weight": "model-00018-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.3.w1.weight": "model-00018-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.2.w3.weight": "model-00018-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.2.w2.weight": "model-00018-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.2.w1.weight": "model-00018-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.1.w3.weight": "model-00018-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.1.w2.weight": "model-00018-of-00048.safetensors", "model.layers.10.block_sparse_moe.experts.1.w1.weight": "model-00018-of-00048.safetensors", "model.layers.10.post_attention_layernorm.weight": "model-00018-of-00048.safetensors", "model.layers.10.input_layernorm.weight": "model-00018-of-00048.safetensors", "model.layers.16.block_sparse_moe.gate.weight": "model-00018-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.1.w2.weight": "model-00019-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.1.w1.weight": "model-00019-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.0.w3.weight": "model-00019-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.0.w2.weight": "model-00019-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.0.w1.weight": "model-00019-of-00048.safetensors", "model.layers.13.self_attn.o_proj.weight": "model-00019-of-00048.safetensors", "model.layers.13.self_attn.v_proj.weight": "model-00019-of-00048.safetensors", "model.layers.13.self_attn.k_proj.weight": "model-00019-of-00048.safetensors", "model.layers.13.self_attn.q_proj.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.gate.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.7.w3.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.7.w2.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.7.w1.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.6.w3.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.6.w2.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.6.w1.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.5.w3.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.5.w2.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.5.w1.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.4.w3.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.4.w2.weight": "model-00019-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.4.w1.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.3.w3.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.3.w2.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.3.w1.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.2.w3.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.2.w2.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.2.w1.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.1.w3.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.1.w2.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.1.w1.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.0.w3.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.0.w2.weight": "model-00020-of-00048.safetensors", "model.layers.12.block_sparse_moe.experts.0.w1.weight": "model-00020-of-00048.safetensors", "model.layers.12.self_attn.o_proj.weight": "model-00020-of-00048.safetensors", "model.layers.12.self_attn.v_proj.weight": "model-00020-of-00048.safetensors", "model.layers.12.self_attn.k_proj.weight": "model-00020-of-00048.safetensors", "model.layers.12.self_attn.q_proj.weight": "model-00020-of-00048.safetensors", "model.layers.12.post_attention_layernorm.weight": "model-00020-of-00048.safetensors", "model.layers.12.input_layernorm.weight": "model-00020-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.7.w3.weight": "model-00020-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.7.w2.weight": "model-00020-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.7.w1.weight": "model-00020-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.1.w3.weight": "model-00021-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.1.w2.weight": "model-00021-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.1.w1.weight": "model-00021-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.0.w3.weight": "model-00021-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.0.w2.weight": "model-00021-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.0.w1.weight": "model-00021-of-00048.safetensors", "model.layers.15.self_attn.o_proj.weight": "model-00021-of-00048.safetensors", "model.layers.15.self_attn.v_proj.weight": "model-00021-of-00048.safetensors", "model.layers.15.self_attn.k_proj.weight": "model-00021-of-00048.safetensors", "model.layers.15.self_attn.q_proj.weight": "model-00021-of-00048.safetensors", "model.layers.14.block_sparse_moe.gate.weight": "model-00021-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.7.w3.weight": "model-00021-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.7.w2.weight": "model-00021-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.7.w1.weight": "model-00021-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.6.w3.weight": "model-00021-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.6.w2.weight": "model-00021-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.6.w1.weight": "model-00021-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.5.w3.weight": "model-00021-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.5.w2.weight": "model-00021-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.5.w1.weight": "model-00021-of-00048.safetensors", "model.layers.11.block_sparse_moe.experts.6.w3.weight": "model-00021-of-00048.safetensors", "model.layers.11.post_attention_layernorm.weight": "model-00021-of-00048.safetensors", "model.layers.11.input_layernorm.weight": "model-00021-of-00048.safetensors", "model.layers.18.block_sparse_moe.gate.weight": "model-00021-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.4.w3.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.4.w2.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.4.w1.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.3.w3.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.3.w2.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.3.w1.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.2.w3.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.2.w2.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.2.w1.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.1.w3.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.1.w2.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.1.w1.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.0.w3.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.0.w2.weight": "model-00022-of-00048.safetensors", "model.layers.14.block_sparse_moe.experts.0.w1.weight": "model-00022-of-00048.safetensors", "model.layers.14.self_attn.o_proj.weight": "model-00022-of-00048.safetensors", "model.layers.14.self_attn.v_proj.weight": "model-00022-of-00048.safetensors", "model.layers.14.self_attn.k_proj.weight": "model-00022-of-00048.safetensors", "model.layers.14.self_attn.q_proj.weight": "model-00022-of-00048.safetensors", "model.layers.14.post_attention_layernorm.weight": "model-00022-of-00048.safetensors", "model.layers.14.input_layernorm.weight": "model-00022-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.7.w3.weight": "model-00022-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.7.w2.weight": "model-00023-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.7.w1.weight": "model-00023-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.6.w3.weight": "model-00023-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.6.w2.weight": "model-00023-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.6.w1.weight": "model-00023-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.5.w3.weight": "model-00023-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.5.w2.weight": "model-00023-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.7.w2.weight": "model-00023-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.7.w1.weight": "model-00023-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.6.w3.weight": "model-00023-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.6.w2.weight": "model-00023-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.6.w1.weight": "model-00023-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.5.w3.weight": "model-00023-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.5.w2.weight": "model-00023-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.5.w1.weight": "model-00023-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.4.w3.weight": "model-00023-of-00048.safetensors", "model.layers.13.block_sparse_moe.experts.4.w2.weight": "model-00023-of-00048.safetensors", "model.layers.13.post_attention_layernorm.weight": "model-00023-of-00048.safetensors", "model.layers.13.input_layernorm.weight": "model-00023-of-00048.safetensors", "model.layers.20.block_sparse_moe.gate.weight": "model-00023-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.5.w1.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.4.w3.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.4.w2.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.4.w1.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.3.w3.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.3.w2.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.3.w1.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.2.w3.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.2.w2.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.2.w1.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.1.w3.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.1.w2.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.1.w1.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.0.w3.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.0.w2.weight": "model-00024-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.0.w1.weight": "model-00024-of-00048.safetensors", "model.layers.16.self_attn.o_proj.weight": "model-00024-of-00048.safetensors", "model.layers.16.self_attn.v_proj.weight": "model-00024-of-00048.safetensors", "model.layers.16.self_attn.k_proj.weight": "model-00024-of-00048.safetensors", "model.layers.16.self_attn.q_proj.weight": "model-00024-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.7.w3.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.7.w2.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.7.w1.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.6.w3.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.6.w2.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.6.w1.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.5.w3.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.5.w2.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.5.w1.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.4.w3.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.4.w2.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.4.w1.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.3.w3.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.3.w2.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.3.w1.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.2.w3.weight": "model-00025-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.2.w2.weight": "model-00025-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.5.w1.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.4.w3.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.4.w2.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.4.w1.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.3.w3.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.3.w2.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.3.w1.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.2.w3.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.2.w2.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.2.w1.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.1.w3.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.1.w2.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.1.w1.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.0.w3.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.0.w2.weight": "model-00026-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.0.w1.weight": "model-00026-of-00048.safetensors", "model.layers.15.block_sparse_moe.experts.2.w1.weight": "model-00026-of-00048.safetensors", "model.layers.15.post_attention_layernorm.weight": "model-00026-of-00048.safetensors", "model.layers.15.input_layernorm.weight": "model-00026-of-00048.safetensors", "model.layers.22.block_sparse_moe.gate.weight": "model-00026-of-00048.safetensors", "model.layers.18.self_attn.o_proj.weight": "model-00027-of-00048.safetensors", "model.layers.18.self_attn.v_proj.weight": "model-00027-of-00048.safetensors", "model.layers.18.self_attn.k_proj.weight": "model-00027-of-00048.safetensors", "model.layers.18.self_attn.q_proj.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.gate.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.7.w3.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.7.w2.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.7.w1.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.6.w3.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.6.w2.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.6.w1.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.5.w3.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.5.w2.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.5.w1.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.4.w3.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.4.w2.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.4.w1.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.3.w3.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.3.w2.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.3.w1.weight": "model-00027-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.2.w3.weight": "model-00027-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.2.w3.weight": "model-00028-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.2.w2.weight": "model-00028-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.2.w1.weight": "model-00028-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.1.w3.weight": "model-00028-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.1.w2.weight": "model-00028-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.1.w1.weight": "model-00028-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.0.w3.weight": "model-00028-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.2.w2.weight": "model-00028-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.2.w1.weight": "model-00028-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.1.w3.weight": "model-00028-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.1.w2.weight": "model-00028-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.1.w1.weight": "model-00028-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.0.w3.weight": "model-00028-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.0.w2.weight": "model-00028-of-00048.safetensors", "model.layers.17.block_sparse_moe.experts.0.w1.weight": "model-00028-of-00048.safetensors", "model.layers.17.self_attn.o_proj.weight": "model-00028-of-00048.safetensors", "model.layers.17.self_attn.v_proj.weight": "model-00028-of-00048.safetensors", "model.layers.17.self_attn.k_proj.weight": "model-00028-of-00048.safetensors", "model.layers.17.self_attn.q_proj.weight": "model-00028-of-00048.safetensors", "model.layers.17.post_attention_layernorm.weight": "model-00028-of-00048.safetensors", "model.layers.17.input_layernorm.weight": "model-00028-of-00048.safetensors", "model.layers.16.block_sparse_moe.experts.7.w3.weight": "model-00028-of-00048.safetensors", "model.layers.16.post_attention_layernorm.weight": "model-00028-of-00048.safetensors", "model.layers.16.input_layernorm.weight": "model-00028-of-00048.safetensors", "model.layers.23.block_sparse_moe.gate.weight": "model-00028-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.0.w2.weight": "model-00029-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.0.w1.weight": "model-00029-of-00048.safetensors", "model.layers.20.self_attn.o_proj.weight": "model-00029-of-00048.safetensors", "model.layers.20.self_attn.v_proj.weight": "model-00029-of-00048.safetensors", "model.layers.20.self_attn.k_proj.weight": "model-00029-of-00048.safetensors", "model.layers.20.self_attn.q_proj.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.gate.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.7.w3.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.7.w2.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.7.w1.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.6.w3.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.6.w2.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.6.w1.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.5.w3.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.5.w2.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.5.w1.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.4.w3.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.4.w2.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.4.w1.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.3.w3.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.3.w2.weight": "model-00029-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.3.w1.weight": "model-00030-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.2.w3.weight": "model-00030-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.2.w2.weight": "model-00030-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.2.w1.weight": "model-00030-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.1.w3.weight": "model-00030-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.1.w2.weight": "model-00030-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.1.w1.weight": "model-00030-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.0.w3.weight": "model-00030-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.0.w2.weight": "model-00030-of-00048.safetensors", "model.layers.19.block_sparse_moe.experts.0.w1.weight": "model-00030-of-00048.safetensors", "model.layers.19.self_attn.o_proj.weight": "model-00030-of-00048.safetensors", "model.layers.19.self_attn.v_proj.weight": "model-00030-of-00048.safetensors", "model.layers.19.self_attn.k_proj.weight": "model-00030-of-00048.safetensors", "model.layers.19.self_attn.q_proj.weight": "model-00030-of-00048.safetensors", "model.layers.19.post_attention_layernorm.weight": "model-00030-of-00048.safetensors", "model.layers.19.input_layernorm.weight": "model-00030-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.7.w3.weight": "model-00030-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.7.w2.weight": "model-00030-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.7.w1.weight": "model-00030-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.6.w3.weight": "model-00030-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.6.w2.weight": "model-00030-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.6.w1.weight": "model-00030-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.0.w2.weight": "model-00031-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.0.w1.weight": "model-00031-of-00048.safetensors", "model.layers.22.self_attn.o_proj.weight": "model-00031-of-00048.safetensors", "model.layers.22.self_attn.v_proj.weight": "model-00031-of-00048.safetensors", "model.layers.22.self_attn.k_proj.weight": "model-00031-of-00048.safetensors", "model.layers.22.self_attn.q_proj.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.gate.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.7.w3.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.7.w2.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.7.w1.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.6.w3.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.6.w2.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.6.w1.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.5.w3.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.5.w2.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.5.w1.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.4.w3.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.4.w2.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.4.w1.weight": "model-00031-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.5.w3.weight": "model-00031-of-00048.safetensors", "model.layers.18.block_sparse_moe.experts.5.w2.weight": "model-00031-of-00048.safetensors", "model.layers.18.post_attention_layernorm.weight": "model-00031-of-00048.safetensors", "model.layers.18.input_layernorm.weight": "model-00031-of-00048.safetensors", "model.layers.25.block_sparse_moe.gate.weight": "model-00031-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.3.w3.weight": "model-00032-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.3.w2.weight": "model-00032-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.3.w1.weight": "model-00032-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.2.w3.weight": "model-00032-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.2.w2.weight": "model-00032-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.2.w1.weight": "model-00032-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.1.w3.weight": "model-00032-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.1.w2.weight": "model-00032-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.1.w1.weight": "model-00032-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.0.w3.weight": "model-00032-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.0.w2.weight": "model-00032-of-00048.safetensors", "model.layers.21.block_sparse_moe.experts.0.w1.weight": "model-00032-of-00048.safetensors", "model.layers.21.self_attn.o_proj.weight": "model-00032-of-00048.safetensors", "model.layers.21.self_attn.v_proj.weight": "model-00032-of-00048.safetensors", "model.layers.21.self_attn.k_proj.weight": "model-00032-of-00048.safetensors", "model.layers.21.self_attn.q_proj.weight": "model-00032-of-00048.safetensors", "model.layers.21.post_attention_layernorm.weight": "model-00032-of-00048.safetensors", "model.layers.21.input_layernorm.weight": "model-00032-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.7.w3.weight": "model-00032-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.7.w2.weight": "model-00032-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.7.w1.weight": "model-00032-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.6.w3.weight": "model-00032-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.6.w1.weight": "model-00033-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.5.w3.weight": "model-00033-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.5.w2.weight": "model-00033-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.5.w1.weight": "model-00033-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.4.w3.weight": "model-00033-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.4.w2.weight": "model-00033-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.6.w2.weight": "model-00033-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.6.w1.weight": "model-00033-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.5.w3.weight": "model-00033-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.5.w2.weight": "model-00033-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.5.w1.weight": "model-00033-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.4.w3.weight": "model-00033-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.4.w2.weight": "model-00033-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.4.w1.weight": "model-00033-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.3.w3.weight": "model-00033-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.3.w2.weight": "model-00033-of-00048.safetensors", "model.layers.20.block_sparse_moe.experts.3.w1.weight": "model-00033-of-00048.safetensors", "model.layers.20.post_attention_layernorm.weight": "model-00033-of-00048.safetensors", "model.layers.20.input_layernorm.weight": "model-00033-of-00048.safetensors", "model.layers.27.block_sparse_moe.gate.weight": "model-00033-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.4.w1.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.3.w3.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.3.w2.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.3.w1.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.2.w3.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.2.w2.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.2.w1.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.1.w3.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.1.w2.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.1.w1.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.0.w3.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.0.w2.weight": "model-00034-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.0.w1.weight": "model-00034-of-00048.safetensors", "model.layers.23.self_attn.o_proj.weight": "model-00034-of-00048.safetensors", "model.layers.23.self_attn.v_proj.weight": "model-00034-of-00048.safetensors", "model.layers.23.self_attn.k_proj.weight": "model-00034-of-00048.safetensors", "model.layers.23.self_attn.q_proj.weight": "model-00034-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.7.w3.weight": "model-00034-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.7.w2.weight": "model-00034-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.7.w1.weight": "model-00034-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.6.w3.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.6.w2.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.6.w1.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.5.w3.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.5.w2.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.5.w1.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.4.w3.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.4.w2.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.4.w1.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.3.w3.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.3.w2.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.3.w1.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.2.w3.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.2.w2.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.2.w1.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.1.w3.weight": "model-00035-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.1.w2.weight": "model-00035-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.3.w3.weight": "model-00036-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.3.w2.weight": "model-00036-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.3.w1.weight": "model-00036-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.2.w3.weight": "model-00036-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.2.w2.weight": "model-00036-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.2.w1.weight": "model-00036-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.1.w3.weight": "model-00036-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.1.w2.weight": "model-00036-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.1.w1.weight": "model-00036-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.0.w3.weight": "model-00036-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.0.w2.weight": "model-00036-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.0.w1.weight": "model-00036-of-00048.safetensors", "model.layers.25.self_attn.o_proj.weight": "model-00036-of-00048.safetensors", "model.layers.25.self_attn.v_proj.weight": "model-00036-of-00048.safetensors", "model.layers.25.self_attn.k_proj.weight": "model-00036-of-00048.safetensors", "model.layers.25.self_attn.q_proj.weight": "model-00036-of-00048.safetensors", "model.layers.24.block_sparse_moe.gate.weight": "model-00036-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.7.w3.weight": "model-00036-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.7.w2.weight": "model-00036-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.1.w1.weight": "model-00036-of-00048.safetensors", "model.layers.22.block_sparse_moe.experts.0.w3.weight": "model-00036-of-00048.safetensors", "model.layers.22.post_attention_layernorm.weight": "model-00036-of-00048.safetensors", "model.layers.22.input_layernorm.weight": "model-00036-of-00048.safetensors", "model.layers.28.block_sparse_moe.gate.weight": "model-00036-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.7.w1.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.6.w3.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.6.w2.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.6.w1.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.5.w3.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.5.w2.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.5.w1.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.4.w3.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.4.w2.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.4.w1.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.3.w3.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.3.w2.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.3.w1.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.2.w3.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.2.w2.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.2.w1.weight": "model-00037-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.1.w3.weight": "model-00037-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.1.w2.weight": "model-00038-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.1.w1.weight": "model-00038-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.0.w3.weight": "model-00038-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.0.w2.weight": "model-00038-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.0.w1.weight": "model-00038-of-00048.safetensors", "model.layers.27.self_attn.o_proj.weight": "model-00038-of-00048.safetensors", "model.layers.27.self_attn.v_proj.weight": "model-00038-of-00048.safetensors", "model.layers.27.self_attn.k_proj.weight": "model-00038-of-00048.safetensors", "model.layers.27.self_attn.q_proj.weight": "model-00038-of-00048.safetensors", "model.layers.26.block_sparse_moe.gate.weight": "model-00038-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.1.w2.weight": "model-00038-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.1.w1.weight": "model-00038-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.0.w3.weight": "model-00038-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.0.w2.weight": "model-00038-of-00048.safetensors", "model.layers.24.block_sparse_moe.experts.0.w1.weight": "model-00038-of-00048.safetensors", "model.layers.24.self_attn.o_proj.weight": "model-00038-of-00048.safetensors", "model.layers.24.self_attn.v_proj.weight": "model-00038-of-00048.safetensors", "model.layers.24.self_attn.k_proj.weight": "model-00038-of-00048.safetensors", "model.layers.24.self_attn.q_proj.weight": "model-00038-of-00048.safetensors", "model.layers.24.post_attention_layernorm.weight": "model-00038-of-00048.safetensors", "model.layers.24.input_layernorm.weight": "model-00038-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.7.w3.weight": "model-00038-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.7.w2.weight": "model-00038-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.7.w1.weight": "model-00038-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.6.w3.weight": "model-00038-of-00048.safetensors", "model.layers.23.block_sparse_moe.experts.6.w2.weight": "model-00038-of-00048.safetensors", "model.layers.23.post_attention_layernorm.weight": "model-00038-of-00048.safetensors", "model.layers.23.input_layernorm.weight": "model-00038-of-00048.safetensors", "model.layers.30.block_sparse_moe.gate.weight": "model-00038-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.7.w3.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.7.w2.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.7.w1.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.6.w3.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.6.w2.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.6.w1.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.5.w3.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.5.w2.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.5.w1.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.4.w3.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.4.w2.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.4.w1.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.3.w3.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.3.w2.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.3.w1.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.2.w3.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.2.w2.weight": "model-00039-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.2.w1.weight": "model-00040-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.1.w3.weight": "model-00040-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.1.w2.weight": "model-00040-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.1.w1.weight": "model-00040-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.0.w3.weight": "model-00040-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.0.w2.weight": "model-00040-of-00048.safetensors", "model.layers.26.block_sparse_moe.experts.0.w1.weight": "model-00040-of-00048.safetensors", "model.layers.26.self_attn.o_proj.weight": "model-00040-of-00048.safetensors", "model.layers.26.self_attn.v_proj.weight": "model-00040-of-00048.safetensors", "model.layers.26.self_attn.k_proj.weight": "model-00040-of-00048.safetensors", "model.layers.26.self_attn.q_proj.weight": "model-00040-of-00048.safetensors", "model.layers.26.post_attention_layernorm.weight": "model-00040-of-00048.safetensors", "model.layers.26.input_layernorm.weight": "model-00040-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.7.w3.weight": "model-00040-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.7.w2.weight": "model-00040-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.7.w1.weight": "model-00040-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.6.w3.weight": "model-00040-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.6.w2.weight": "model-00040-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.6.w1.weight": "model-00040-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.5.w3.weight": "model-00040-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.5.w2.weight": "model-00040-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.5.w1.weight": "model-00040-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.7.w1.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.6.w3.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.6.w2.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.6.w1.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.5.w3.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.5.w2.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.5.w1.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.4.w3.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.4.w2.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.4.w1.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.3.w3.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.3.w2.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.3.w1.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.2.w3.weight": "model-00041-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.4.w3.weight": "model-00041-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.4.w2.weight": "model-00041-of-00048.safetensors", "model.layers.25.block_sparse_moe.experts.4.w1.weight": "model-00041-of-00048.safetensors", "model.layers.25.post_attention_layernorm.weight": "model-00041-of-00048.safetensors", "model.layers.25.input_layernorm.weight": "model-00041-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.2.w2.weight": "model-00042-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.2.w1.weight": "model-00042-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.1.w3.weight": "model-00042-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.1.w2.weight": "model-00042-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.1.w1.weight": "model-00042-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.0.w3.weight": "model-00042-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.0.w2.weight": "model-00042-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.0.w1.weight": "model-00042-of-00048.safetensors", "model.layers.28.self_attn.o_proj.weight": "model-00042-of-00048.safetensors", "model.layers.28.self_attn.v_proj.weight": "model-00042-of-00048.safetensors", "model.layers.28.self_attn.k_proj.weight": "model-00042-of-00048.safetensors", "model.layers.28.self_attn.q_proj.weight": "model-00042-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.7.w3.weight": "model-00042-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.7.w2.weight": "model-00042-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.7.w1.weight": "model-00042-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.6.w3.weight": "model-00042-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.6.w2.weight": "model-00042-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.6.w1.weight": "model-00042-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.5.w3.weight": "model-00042-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.5.w2.weight": "model-00042-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.4.w3.weight": "model-00043-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.4.w2.weight": "model-00043-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.4.w1.weight": "model-00043-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.3.w3.weight": "model-00043-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.3.w2.weight": "model-00043-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.3.w1.weight": "model-00043-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.5.w1.weight": "model-00043-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.4.w3.weight": "model-00043-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.4.w2.weight": "model-00043-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.4.w1.weight": "model-00043-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.3.w3.weight": "model-00043-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.3.w2.weight": "model-00043-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.3.w1.weight": "model-00043-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.2.w3.weight": "model-00043-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.2.w2.weight": "model-00043-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.2.w1.weight": "model-00043-of-00048.safetensors", "model.layers.27.block_sparse_moe.experts.1.w3.weight": "model-00043-of-00048.safetensors", "model.layers.27.post_attention_layernorm.weight": "model-00043-of-00048.safetensors", "model.layers.27.input_layernorm.weight": "model-00043-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.2.w3.weight": "model-00044-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.2.w2.weight": "model-00044-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.2.w1.weight": "model-00044-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.1.w3.weight": "model-00044-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.1.w2.weight": "model-00044-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.1.w1.weight": "model-00044-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.0.w3.weight": "model-00044-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.0.w2.weight": "model-00044-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.0.w1.weight": "model-00044-of-00048.safetensors", "model.layers.30.self_attn.o_proj.weight": "model-00044-of-00048.safetensors", "model.layers.30.self_attn.v_proj.weight": "model-00044-of-00048.safetensors", "model.layers.30.self_attn.k_proj.weight": "model-00044-of-00048.safetensors", "model.layers.30.self_attn.q_proj.weight": "model-00044-of-00048.safetensors", "model.layers.29.block_sparse_moe.gate.weight": "model-00044-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.7.w3.weight": "model-00044-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.7.w2.weight": "model-00044-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.7.w1.weight": "model-00044-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.6.w3.weight": "model-00044-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.6.w2.weight": "model-00044-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.6.w1.weight": "model-00044-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.5.w3.weight": "model-00044-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.5.w2.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.5.w1.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.4.w3.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.4.w2.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.4.w1.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.3.w3.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.3.w2.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.3.w1.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.2.w3.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.2.w2.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.2.w1.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.1.w3.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.1.w2.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.1.w1.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.0.w3.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.0.w2.weight": "model-00045-of-00048.safetensors", "model.layers.29.block_sparse_moe.experts.0.w1.weight": "model-00045-of-00048.safetensors", "lm_head.weight": "model-00046-of-00048.safetensors", "model.norm.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.gate.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.7.w3.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.7.w2.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.7.w1.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.6.w3.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.6.w2.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.6.w1.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.5.w3.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.5.w2.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.5.w1.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.4.w3.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.4.w2.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.4.w1.weight": "model-00046-of-00048.safetensors", "model.layers.29.self_attn.o_proj.weight": "model-00046-of-00048.safetensors", "model.layers.29.self_attn.v_proj.weight": "model-00046-of-00048.safetensors", "model.layers.29.self_attn.k_proj.weight": "model-00046-of-00048.safetensors", "model.layers.29.self_attn.q_proj.weight": "model-00046-of-00048.safetensors", "model.layers.29.post_attention_layernorm.weight": "model-00046-of-00048.safetensors", "model.layers.29.input_layernorm.weight": "model-00046-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.7.w3.weight": "model-00046-of-00048.safetensors", "model.layers.28.block_sparse_moe.experts.7.w2.weight": "model-00046-of-00048.safetensors", "model.layers.28.post_attention_layernorm.weight": "model-00046-of-00048.safetensors", "model.layers.28.input_layernorm.weight": "model-00046-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.3.w3.weight": "model-00047-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.3.w2.weight": "model-00047-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.3.w1.weight": "model-00047-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.2.w3.weight": "model-00047-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.2.w2.weight": "model-00047-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.2.w1.weight": "model-00047-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.1.w3.weight": "model-00047-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.1.w2.weight": "model-00047-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.1.w1.weight": "model-00047-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.0.w3.weight": "model-00047-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.0.w2.weight": "model-00047-of-00048.safetensors", "model.layers.31.block_sparse_moe.experts.0.w1.weight": "model-00047-of-00048.safetensors", "model.layers.31.self_attn.o_proj.weight": "model-00047-of-00048.safetensors", "model.layers.31.self_attn.v_proj.weight": "model-00047-of-00048.safetensors", "model.layers.31.self_attn.k_proj.weight": "model-00047-of-00048.safetensors", "model.layers.31.self_attn.q_proj.weight": "model-00047-of-00048.safetensors", "model.layers.31.post_attention_layernorm.weight": "model-00047-of-00048.safetensors", "model.layers.31.input_layernorm.weight": "model-00047-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.7.w3.weight": "model-00047-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.7.w2.weight": "model-00047-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.7.w1.weight": "model-00047-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.6.w3.weight": "model-00047-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.6.w2.weight": "model-00048-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.6.w1.weight": "model-00048-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.5.w3.weight": "model-00048-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.5.w2.weight": "model-00048-of-00048.safetensors", "model.layers.30.block_sparse_moe.experts.5.w1.weight": "model-00048-of-00048.safetensors", "model.layers.30.post_attention_layernorm.weight": "model-00048-of-00048.safetensors", "model.layers.30.input_layernorm.weight": "model-00048-of-00048.safetensors"}}
 
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/output-00001-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:d5016af02014c6a77dd99f08051e9c348eaf58517ad5884bd55d003e14b93ab6
3
- size 8590133528
 
 
 
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/output-00002-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:8d557fe5f014eb2708d3957215c6ac7f6f6223cb077fa2db6e2e2c0b47ab2483
3
- size 8590083224
 
 
 
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/output-00003-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:e90a0a4ba13d88a43bfeb9fc81eaaf1568529794e8150515275c8e3fd4a5f4a0
3
- size 8549890824
 
 
 
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/output-00004-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:758cec69021174e8ee23e055655ec755893f406380d68283359fcab10479f7c7
3
- size 3663577200
 
 
 
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/special_tokens_map.json DELETED
@@ -1,23 +0,0 @@
1
- {
2
- "bos_token": {
3
- "content": "<s>",
4
- "lstrip": false,
5
- "normalized": false,
6
- "rstrip": false,
7
- "single_word": false
8
- },
9
- "eos_token": {
10
- "content": "</s>",
11
- "lstrip": false,
12
- "normalized": false,
13
- "rstrip": false,
14
- "single_word": false
15
- },
16
- "unk_token": {
17
- "content": "<unk>",
18
- "lstrip": false,
19
- "normalized": false,
20
- "rstrip": false,
21
- "single_word": false
22
- }
23
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/tokenizer.json DELETED
The diff for this file is too large to render. See raw diff
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/tokenizer.model DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:dadfd56d766715c61d2ef780a525ab43b8e6da4de6865bda3d95fdef5e134055
3
- size 493443
 
 
 
 
https:/huggingface.co/maeeeeee/maid-yuzu-v8-alter-5.0bpw-exl2/tree/main/tokenizer_config.json DELETED
@@ -1,42 +0,0 @@
1
- {
2
- "add_bos_token": true,
3
- "add_eos_token": false,
4
- "added_tokens_decoder": {
5
- "0": {
6
- "content": "<unk>",
7
- "lstrip": false,
8
- "normalized": false,
9
- "rstrip": false,
10
- "single_word": false,
11
- "special": true
12
- },
13
- "1": {
14
- "content": "<s>",
15
- "lstrip": false,
16
- "normalized": false,
17
- "rstrip": false,
18
- "single_word": false,
19
- "special": true
20
- },
21
- "2": {
22
- "content": "</s>",
23
- "lstrip": false,
24
- "normalized": false,
25
- "rstrip": false,
26
- "single_word": false,
27
- "special": true
28
- }
29
- },
30
- "additional_special_tokens": [],
31
- "bos_token": "<s>",
32
- "clean_up_tokenization_spaces": false,
33
- "eos_token": "</s>",
34
- "legacy": true,
35
- "model_max_length": 1000000000000000019884624838656,
36
- "pad_token": null,
37
- "sp_model_kwargs": {},
38
- "spaces_between_special_tokens": false,
39
- "tokenizer_class": "LlamaTokenizer",
40
- "unk_token": "<unk>",
41
- "use_default_system_prompt": false
42
- }