Darkhn commited on
Commit
8977d84
·
verified ·
1 Parent(s): 92fdb21

Upload 12 files

Browse files
.gitattributes CHANGED
@@ -1,35 +1,36 @@
1
- *.7z filter=lfs diff=lfs merge=lfs -text
2
- *.arrow filter=lfs diff=lfs merge=lfs -text
3
- *.bin filter=lfs diff=lfs merge=lfs -text
4
- *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
- *.ftz filter=lfs diff=lfs merge=lfs -text
7
- *.gz filter=lfs diff=lfs merge=lfs -text
8
- *.h5 filter=lfs diff=lfs merge=lfs -text
9
- *.joblib filter=lfs diff=lfs merge=lfs -text
10
- *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
- *.model filter=lfs diff=lfs merge=lfs -text
13
- *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
- *.onnx filter=lfs diff=lfs merge=lfs -text
17
- *.ot filter=lfs diff=lfs merge=lfs -text
18
- *.parquet filter=lfs diff=lfs merge=lfs -text
19
- *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
- *.pt filter=lfs diff=lfs merge=lfs -text
23
- *.pth filter=lfs diff=lfs merge=lfs -text
24
- *.rar filter=lfs diff=lfs merge=lfs -text
25
- *.safetensors filter=lfs diff=lfs merge=lfs -text
26
- saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
- *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
- *.tflite filter=lfs diff=lfs merge=lfs -text
30
- *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
- *.xz filter=lfs diff=lfs merge=lfs -text
33
- *.zip filter=lfs diff=lfs merge=lfs -text
34
- *.zst filter=lfs diff=lfs merge=lfs -text
35
- *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
README.md CHANGED
@@ -1,3 +1,51 @@
1
  ---
 
 
 
 
 
 
 
2
  license: apache-2.0
 
 
3
  ---
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  ---
2
+ base_model: mistralai/Mistral-Small-3.1-24B-Instruct-2503
3
+ tags:
4
+ - text-generation-inference
5
+ - transformers
6
+ - unsloth
7
+ - mistral
8
+ - trl
9
  license: apache-2.0
10
+ language:
11
+ - en
12
  ---
13
+
14
+ ![Eurydice 24b Banner](https://cdn-uploads.huggingface.co/production/uploads/652c2a63d78452c4742cd3d3/Hm_tg4s0D6yWmtrTHII32.png)
15
+
16
+ # Eurydice 24b v2 🧙‍♂️
17
+
18
+
19
+ Eurydice 24b v2 is designed to be the perfect companion for multi-role conversations. It demonstrates exceptional contextual understanding and excels in creativity, natural conversation and storytelling. Built on Mistral 3.1, this model has been trained on a custom dataset specifically crafted to enhance its capabilities.
20
+
21
+ ## Model Details 📊
22
+
23
+ - **Developed by:** Aixon Lab
24
+ - **Model type:** Causal Language Model
25
+ - **Language(s):** English (primarily), may support other languages
26
+ - **License:** Apache 2.0
27
+ - **Repository:** https://huggingface.co/aixonlab/Eurydice-24b-v2
28
+
29
+ ## Quantization
30
+ - **GGUF:** Coming Soon !
31
+
32
+ ## Model Architecture 🏗️
33
+
34
+ - **Base model:** mistralai/Mistral-Small-3.1-24B-Instruct-2503
35
+ - **Parameter count:** ~24 billion
36
+ - **Architecture specifics:** Transformer-based language model
37
+
38
+ ## Intended Use 🎯
39
+ As an advanced language model for various natural language processing tasks, including but not limited to text generation (excels in chat), question-answering, and analysis.
40
+
41
+ ## Ethical Considerations 🤔
42
+ As a model based on multiple sources, Eurydice 24b v2 may inherit biases and limitations from its constituent models. Users should be aware of potential biases in generated content and use the model responsibly.
43
+
44
+ ## Performance and Evaluation
45
+ Performance metrics and evaluation results for Eurydice 24b v2 are yet to be determined. Users are encouraged to contribute their findings and benchmarks.
46
+
47
+ ## Limitations and Biases
48
+ The model may exhibit biases present in its training data and constituent models. It's crucial to critically evaluate the model's outputs and use them in conjunction with human judgment.
49
+
50
+ ## Additional Information
51
+ For more details on the base model and constituent models, please refer to their respective model cards and documentation.
config.json ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "MistralForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": 1,
7
+ "eos_token_id": 999,
8
+ "head_dim": 128,
9
+ "hidden_act": "silu",
10
+ "hidden_size": 5120,
11
+ "initializer_range": 0.02,
12
+ "intermediate_size": 32768,
13
+ "max_position_embeddings": 131072,
14
+ "model_type": "mistral",
15
+ "num_attention_heads": 32,
16
+ "num_hidden_layers": 40,
17
+ "num_key_value_heads": 8,
18
+ "pad_token_id": 11,
19
+ "rms_norm_eps": 1e-05,
20
+ "rope_scaling": null,
21
+ "rope_theta": 1000000000.0,
22
+ "sliding_window": null,
23
+ "tie_word_embeddings": false,
24
+ "torch_dtype": "bfloat16",
25
+ "transformers_version": "4.51.0",
26
+ "unsloth_version": "2025.3.19",
27
+ "use_cache": false,
28
+ "vocab_size": 131072,
29
+ "quantization_config": {
30
+ "quant_method": "exl2",
31
+ "version": "0.2.7",
32
+ "bits": 6.0,
33
+ "head_bits": 8,
34
+ "calibration": {
35
+ "rows": 115,
36
+ "length": 2048,
37
+ "dataset": "(default)"
38
+ }
39
+ }
40
+ }
generation_config.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "bos_token_id": 1,
4
+ "do_sample": true,
5
+ "eos_token_id": 2,
6
+ "max_length": 131072,
7
+ "pad_token_id": 11,
8
+ "transformers_version": "4.51.0"
9
+ }
measurement.json ADDED
The diff for this file is too large to render. See raw diff
 
model.safetensors.index.json ADDED
@@ -0,0 +1,370 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "total_size": 47144806400
4
+ },
5
+ "weight_map": {
6
+ "lm_head.weight": "model-00010-of-00010.safetensors",
7
+ "model.embed_tokens.weight": "model-00001-of-00010.safetensors",
8
+ "model.layers.0.input_layernorm.weight": "model-00001-of-00010.safetensors",
9
+ "model.layers.0.mlp.down_proj.weight": "model-00001-of-00010.safetensors",
10
+ "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00010.safetensors",
11
+ "model.layers.0.mlp.up_proj.weight": "model-00001-of-00010.safetensors",
12
+ "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00010.safetensors",
13
+ "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00010.safetensors",
14
+ "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00010.safetensors",
15
+ "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00010.safetensors",
16
+ "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00010.safetensors",
17
+ "model.layers.1.input_layernorm.weight": "model-00001-of-00010.safetensors",
18
+ "model.layers.1.mlp.down_proj.weight": "model-00001-of-00010.safetensors",
19
+ "model.layers.1.mlp.gate_proj.weight": "model-00001-of-00010.safetensors",
20
+ "model.layers.1.mlp.up_proj.weight": "model-00001-of-00010.safetensors",
21
+ "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00010.safetensors",
22
+ "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00010.safetensors",
23
+ "model.layers.1.self_attn.o_proj.weight": "model-00001-of-00010.safetensors",
24
+ "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00010.safetensors",
25
+ "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00010.safetensors",
26
+ "model.layers.10.input_layernorm.weight": "model-00003-of-00010.safetensors",
27
+ "model.layers.10.mlp.down_proj.weight": "model-00003-of-00010.safetensors",
28
+ "model.layers.10.mlp.gate_proj.weight": "model-00003-of-00010.safetensors",
29
+ "model.layers.10.mlp.up_proj.weight": "model-00003-of-00010.safetensors",
30
+ "model.layers.10.post_attention_layernorm.weight": "model-00003-of-00010.safetensors",
31
+ "model.layers.10.self_attn.k_proj.weight": "model-00003-of-00010.safetensors",
32
+ "model.layers.10.self_attn.o_proj.weight": "model-00003-of-00010.safetensors",
33
+ "model.layers.10.self_attn.q_proj.weight": "model-00003-of-00010.safetensors",
34
+ "model.layers.10.self_attn.v_proj.weight": "model-00003-of-00010.safetensors",
35
+ "model.layers.11.input_layernorm.weight": "model-00004-of-00010.safetensors",
36
+ "model.layers.11.mlp.down_proj.weight": "model-00004-of-00010.safetensors",
37
+ "model.layers.11.mlp.gate_proj.weight": "model-00003-of-00010.safetensors",
38
+ "model.layers.11.mlp.up_proj.weight": "model-00003-of-00010.safetensors",
39
+ "model.layers.11.post_attention_layernorm.weight": "model-00004-of-00010.safetensors",
40
+ "model.layers.11.self_attn.k_proj.weight": "model-00003-of-00010.safetensors",
41
+ "model.layers.11.self_attn.o_proj.weight": "model-00003-of-00010.safetensors",
42
+ "model.layers.11.self_attn.q_proj.weight": "model-00003-of-00010.safetensors",
43
+ "model.layers.11.self_attn.v_proj.weight": "model-00003-of-00010.safetensors",
44
+ "model.layers.12.input_layernorm.weight": "model-00004-of-00010.safetensors",
45
+ "model.layers.12.mlp.down_proj.weight": "model-00004-of-00010.safetensors",
46
+ "model.layers.12.mlp.gate_proj.weight": "model-00004-of-00010.safetensors",
47
+ "model.layers.12.mlp.up_proj.weight": "model-00004-of-00010.safetensors",
48
+ "model.layers.12.post_attention_layernorm.weight": "model-00004-of-00010.safetensors",
49
+ "model.layers.12.self_attn.k_proj.weight": "model-00004-of-00010.safetensors",
50
+ "model.layers.12.self_attn.o_proj.weight": "model-00004-of-00010.safetensors",
51
+ "model.layers.12.self_attn.q_proj.weight": "model-00004-of-00010.safetensors",
52
+ "model.layers.12.self_attn.v_proj.weight": "model-00004-of-00010.safetensors",
53
+ "model.layers.13.input_layernorm.weight": "model-00004-of-00010.safetensors",
54
+ "model.layers.13.mlp.down_proj.weight": "model-00004-of-00010.safetensors",
55
+ "model.layers.13.mlp.gate_proj.weight": "model-00004-of-00010.safetensors",
56
+ "model.layers.13.mlp.up_proj.weight": "model-00004-of-00010.safetensors",
57
+ "model.layers.13.post_attention_layernorm.weight": "model-00004-of-00010.safetensors",
58
+ "model.layers.13.self_attn.k_proj.weight": "model-00004-of-00010.safetensors",
59
+ "model.layers.13.self_attn.o_proj.weight": "model-00004-of-00010.safetensors",
60
+ "model.layers.13.self_attn.q_proj.weight": "model-00004-of-00010.safetensors",
61
+ "model.layers.13.self_attn.v_proj.weight": "model-00004-of-00010.safetensors",
62
+ "model.layers.14.input_layernorm.weight": "model-00004-of-00010.safetensors",
63
+ "model.layers.14.mlp.down_proj.weight": "model-00004-of-00010.safetensors",
64
+ "model.layers.14.mlp.gate_proj.weight": "model-00004-of-00010.safetensors",
65
+ "model.layers.14.mlp.up_proj.weight": "model-00004-of-00010.safetensors",
66
+ "model.layers.14.post_attention_layernorm.weight": "model-00004-of-00010.safetensors",
67
+ "model.layers.14.self_attn.k_proj.weight": "model-00004-of-00010.safetensors",
68
+ "model.layers.14.self_attn.o_proj.weight": "model-00004-of-00010.safetensors",
69
+ "model.layers.14.self_attn.q_proj.weight": "model-00004-of-00010.safetensors",
70
+ "model.layers.14.self_attn.v_proj.weight": "model-00004-of-00010.safetensors",
71
+ "model.layers.15.input_layernorm.weight": "model-00004-of-00010.safetensors",
72
+ "model.layers.15.mlp.down_proj.weight": "model-00004-of-00010.safetensors",
73
+ "model.layers.15.mlp.gate_proj.weight": "model-00004-of-00010.safetensors",
74
+ "model.layers.15.mlp.up_proj.weight": "model-00004-of-00010.safetensors",
75
+ "model.layers.15.post_attention_layernorm.weight": "model-00004-of-00010.safetensors",
76
+ "model.layers.15.self_attn.k_proj.weight": "model-00004-of-00010.safetensors",
77
+ "model.layers.15.self_attn.o_proj.weight": "model-00004-of-00010.safetensors",
78
+ "model.layers.15.self_attn.q_proj.weight": "model-00004-of-00010.safetensors",
79
+ "model.layers.15.self_attn.v_proj.weight": "model-00004-of-00010.safetensors",
80
+ "model.layers.16.input_layernorm.weight": "model-00005-of-00010.safetensors",
81
+ "model.layers.16.mlp.down_proj.weight": "model-00005-of-00010.safetensors",
82
+ "model.layers.16.mlp.gate_proj.weight": "model-00005-of-00010.safetensors",
83
+ "model.layers.16.mlp.up_proj.weight": "model-00005-of-00010.safetensors",
84
+ "model.layers.16.post_attention_layernorm.weight": "model-00005-of-00010.safetensors",
85
+ "model.layers.16.self_attn.k_proj.weight": "model-00004-of-00010.safetensors",
86
+ "model.layers.16.self_attn.o_proj.weight": "model-00004-of-00010.safetensors",
87
+ "model.layers.16.self_attn.q_proj.weight": "model-00004-of-00010.safetensors",
88
+ "model.layers.16.self_attn.v_proj.weight": "model-00004-of-00010.safetensors",
89
+ "model.layers.17.input_layernorm.weight": "model-00005-of-00010.safetensors",
90
+ "model.layers.17.mlp.down_proj.weight": "model-00005-of-00010.safetensors",
91
+ "model.layers.17.mlp.gate_proj.weight": "model-00005-of-00010.safetensors",
92
+ "model.layers.17.mlp.up_proj.weight": "model-00005-of-00010.safetensors",
93
+ "model.layers.17.post_attention_layernorm.weight": "model-00005-of-00010.safetensors",
94
+ "model.layers.17.self_attn.k_proj.weight": "model-00005-of-00010.safetensors",
95
+ "model.layers.17.self_attn.o_proj.weight": "model-00005-of-00010.safetensors",
96
+ "model.layers.17.self_attn.q_proj.weight": "model-00005-of-00010.safetensors",
97
+ "model.layers.17.self_attn.v_proj.weight": "model-00005-of-00010.safetensors",
98
+ "model.layers.18.input_layernorm.weight": "model-00005-of-00010.safetensors",
99
+ "model.layers.18.mlp.down_proj.weight": "model-00005-of-00010.safetensors",
100
+ "model.layers.18.mlp.gate_proj.weight": "model-00005-of-00010.safetensors",
101
+ "model.layers.18.mlp.up_proj.weight": "model-00005-of-00010.safetensors",
102
+ "model.layers.18.post_attention_layernorm.weight": "model-00005-of-00010.safetensors",
103
+ "model.layers.18.self_attn.k_proj.weight": "model-00005-of-00010.safetensors",
104
+ "model.layers.18.self_attn.o_proj.weight": "model-00005-of-00010.safetensors",
105
+ "model.layers.18.self_attn.q_proj.weight": "model-00005-of-00010.safetensors",
106
+ "model.layers.18.self_attn.v_proj.weight": "model-00005-of-00010.safetensors",
107
+ "model.layers.19.input_layernorm.weight": "model-00005-of-00010.safetensors",
108
+ "model.layers.19.mlp.down_proj.weight": "model-00005-of-00010.safetensors",
109
+ "model.layers.19.mlp.gate_proj.weight": "model-00005-of-00010.safetensors",
110
+ "model.layers.19.mlp.up_proj.weight": "model-00005-of-00010.safetensors",
111
+ "model.layers.19.post_attention_layernorm.weight": "model-00005-of-00010.safetensors",
112
+ "model.layers.19.self_attn.k_proj.weight": "model-00005-of-00010.safetensors",
113
+ "model.layers.19.self_attn.o_proj.weight": "model-00005-of-00010.safetensors",
114
+ "model.layers.19.self_attn.q_proj.weight": "model-00005-of-00010.safetensors",
115
+ "model.layers.19.self_attn.v_proj.weight": "model-00005-of-00010.safetensors",
116
+ "model.layers.2.input_layernorm.weight": "model-00001-of-00010.safetensors",
117
+ "model.layers.2.mlp.down_proj.weight": "model-00001-of-00010.safetensors",
118
+ "model.layers.2.mlp.gate_proj.weight": "model-00001-of-00010.safetensors",
119
+ "model.layers.2.mlp.up_proj.weight": "model-00001-of-00010.safetensors",
120
+ "model.layers.2.post_attention_layernorm.weight": "model-00001-of-00010.safetensors",
121
+ "model.layers.2.self_attn.k_proj.weight": "model-00001-of-00010.safetensors",
122
+ "model.layers.2.self_attn.o_proj.weight": "model-00001-of-00010.safetensors",
123
+ "model.layers.2.self_attn.q_proj.weight": "model-00001-of-00010.safetensors",
124
+ "model.layers.2.self_attn.v_proj.weight": "model-00001-of-00010.safetensors",
125
+ "model.layers.20.input_layernorm.weight": "model-00006-of-00010.safetensors",
126
+ "model.layers.20.mlp.down_proj.weight": "model-00006-of-00010.safetensors",
127
+ "model.layers.20.mlp.gate_proj.weight": "model-00005-of-00010.safetensors",
128
+ "model.layers.20.mlp.up_proj.weight": "model-00006-of-00010.safetensors",
129
+ "model.layers.20.post_attention_layernorm.weight": "model-00006-of-00010.safetensors",
130
+ "model.layers.20.self_attn.k_proj.weight": "model-00005-of-00010.safetensors",
131
+ "model.layers.20.self_attn.o_proj.weight": "model-00005-of-00010.safetensors",
132
+ "model.layers.20.self_attn.q_proj.weight": "model-00005-of-00010.safetensors",
133
+ "model.layers.20.self_attn.v_proj.weight": "model-00005-of-00010.safetensors",
134
+ "model.layers.21.input_layernorm.weight": "model-00006-of-00010.safetensors",
135
+ "model.layers.21.mlp.down_proj.weight": "model-00006-of-00010.safetensors",
136
+ "model.layers.21.mlp.gate_proj.weight": "model-00006-of-00010.safetensors",
137
+ "model.layers.21.mlp.up_proj.weight": "model-00006-of-00010.safetensors",
138
+ "model.layers.21.post_attention_layernorm.weight": "model-00006-of-00010.safetensors",
139
+ "model.layers.21.self_attn.k_proj.weight": "model-00006-of-00010.safetensors",
140
+ "model.layers.21.self_attn.o_proj.weight": "model-00006-of-00010.safetensors",
141
+ "model.layers.21.self_attn.q_proj.weight": "model-00006-of-00010.safetensors",
142
+ "model.layers.21.self_attn.v_proj.weight": "model-00006-of-00010.safetensors",
143
+ "model.layers.22.input_layernorm.weight": "model-00006-of-00010.safetensors",
144
+ "model.layers.22.mlp.down_proj.weight": "model-00006-of-00010.safetensors",
145
+ "model.layers.22.mlp.gate_proj.weight": "model-00006-of-00010.safetensors",
146
+ "model.layers.22.mlp.up_proj.weight": "model-00006-of-00010.safetensors",
147
+ "model.layers.22.post_attention_layernorm.weight": "model-00006-of-00010.safetensors",
148
+ "model.layers.22.self_attn.k_proj.weight": "model-00006-of-00010.safetensors",
149
+ "model.layers.22.self_attn.o_proj.weight": "model-00006-of-00010.safetensors",
150
+ "model.layers.22.self_attn.q_proj.weight": "model-00006-of-00010.safetensors",
151
+ "model.layers.22.self_attn.v_proj.weight": "model-00006-of-00010.safetensors",
152
+ "model.layers.23.input_layernorm.weight": "model-00006-of-00010.safetensors",
153
+ "model.layers.23.mlp.down_proj.weight": "model-00006-of-00010.safetensors",
154
+ "model.layers.23.mlp.gate_proj.weight": "model-00006-of-00010.safetensors",
155
+ "model.layers.23.mlp.up_proj.weight": "model-00006-of-00010.safetensors",
156
+ "model.layers.23.post_attention_layernorm.weight": "model-00006-of-00010.safetensors",
157
+ "model.layers.23.self_attn.k_proj.weight": "model-00006-of-00010.safetensors",
158
+ "model.layers.23.self_attn.o_proj.weight": "model-00006-of-00010.safetensors",
159
+ "model.layers.23.self_attn.q_proj.weight": "model-00006-of-00010.safetensors",
160
+ "model.layers.23.self_attn.v_proj.weight": "model-00006-of-00010.safetensors",
161
+ "model.layers.24.input_layernorm.weight": "model-00007-of-00010.safetensors",
162
+ "model.layers.24.mlp.down_proj.weight": "model-00007-of-00010.safetensors",
163
+ "model.layers.24.mlp.gate_proj.weight": "model-00006-of-00010.safetensors",
164
+ "model.layers.24.mlp.up_proj.weight": "model-00006-of-00010.safetensors",
165
+ "model.layers.24.post_attention_layernorm.weight": "model-00007-of-00010.safetensors",
166
+ "model.layers.24.self_attn.k_proj.weight": "model-00006-of-00010.safetensors",
167
+ "model.layers.24.self_attn.o_proj.weight": "model-00006-of-00010.safetensors",
168
+ "model.layers.24.self_attn.q_proj.weight": "model-00006-of-00010.safetensors",
169
+ "model.layers.24.self_attn.v_proj.weight": "model-00006-of-00010.safetensors",
170
+ "model.layers.25.input_layernorm.weight": "model-00007-of-00010.safetensors",
171
+ "model.layers.25.mlp.down_proj.weight": "model-00007-of-00010.safetensors",
172
+ "model.layers.25.mlp.gate_proj.weight": "model-00007-of-00010.safetensors",
173
+ "model.layers.25.mlp.up_proj.weight": "model-00007-of-00010.safetensors",
174
+ "model.layers.25.post_attention_layernorm.weight": "model-00007-of-00010.safetensors",
175
+ "model.layers.25.self_attn.k_proj.weight": "model-00007-of-00010.safetensors",
176
+ "model.layers.25.self_attn.o_proj.weight": "model-00007-of-00010.safetensors",
177
+ "model.layers.25.self_attn.q_proj.weight": "model-00007-of-00010.safetensors",
178
+ "model.layers.25.self_attn.v_proj.weight": "model-00007-of-00010.safetensors",
179
+ "model.layers.26.input_layernorm.weight": "model-00007-of-00010.safetensors",
180
+ "model.layers.26.mlp.down_proj.weight": "model-00007-of-00010.safetensors",
181
+ "model.layers.26.mlp.gate_proj.weight": "model-00007-of-00010.safetensors",
182
+ "model.layers.26.mlp.up_proj.weight": "model-00007-of-00010.safetensors",
183
+ "model.layers.26.post_attention_layernorm.weight": "model-00007-of-00010.safetensors",
184
+ "model.layers.26.self_attn.k_proj.weight": "model-00007-of-00010.safetensors",
185
+ "model.layers.26.self_attn.o_proj.weight": "model-00007-of-00010.safetensors",
186
+ "model.layers.26.self_attn.q_proj.weight": "model-00007-of-00010.safetensors",
187
+ "model.layers.26.self_attn.v_proj.weight": "model-00007-of-00010.safetensors",
188
+ "model.layers.27.input_layernorm.weight": "model-00007-of-00010.safetensors",
189
+ "model.layers.27.mlp.down_proj.weight": "model-00007-of-00010.safetensors",
190
+ "model.layers.27.mlp.gate_proj.weight": "model-00007-of-00010.safetensors",
191
+ "model.layers.27.mlp.up_proj.weight": "model-00007-of-00010.safetensors",
192
+ "model.layers.27.post_attention_layernorm.weight": "model-00007-of-00010.safetensors",
193
+ "model.layers.27.self_attn.k_proj.weight": "model-00007-of-00010.safetensors",
194
+ "model.layers.27.self_attn.o_proj.weight": "model-00007-of-00010.safetensors",
195
+ "model.layers.27.self_attn.q_proj.weight": "model-00007-of-00010.safetensors",
196
+ "model.layers.27.self_attn.v_proj.weight": "model-00007-of-00010.safetensors",
197
+ "model.layers.28.input_layernorm.weight": "model-00007-of-00010.safetensors",
198
+ "model.layers.28.mlp.down_proj.weight": "model-00007-of-00010.safetensors",
199
+ "model.layers.28.mlp.gate_proj.weight": "model-00007-of-00010.safetensors",
200
+ "model.layers.28.mlp.up_proj.weight": "model-00007-of-00010.safetensors",
201
+ "model.layers.28.post_attention_layernorm.weight": "model-00007-of-00010.safetensors",
202
+ "model.layers.28.self_attn.k_proj.weight": "model-00007-of-00010.safetensors",
203
+ "model.layers.28.self_attn.o_proj.weight": "model-00007-of-00010.safetensors",
204
+ "model.layers.28.self_attn.q_proj.weight": "model-00007-of-00010.safetensors",
205
+ "model.layers.28.self_attn.v_proj.weight": "model-00007-of-00010.safetensors",
206
+ "model.layers.29.input_layernorm.weight": "model-00008-of-00010.safetensors",
207
+ "model.layers.29.mlp.down_proj.weight": "model-00008-of-00010.safetensors",
208
+ "model.layers.29.mlp.gate_proj.weight": "model-00008-of-00010.safetensors",
209
+ "model.layers.29.mlp.up_proj.weight": "model-00008-of-00010.safetensors",
210
+ "model.layers.29.post_attention_layernorm.weight": "model-00008-of-00010.safetensors",
211
+ "model.layers.29.self_attn.k_proj.weight": "model-00007-of-00010.safetensors",
212
+ "model.layers.29.self_attn.o_proj.weight": "model-00007-of-00010.safetensors",
213
+ "model.layers.29.self_attn.q_proj.weight": "model-00007-of-00010.safetensors",
214
+ "model.layers.29.self_attn.v_proj.weight": "model-00007-of-00010.safetensors",
215
+ "model.layers.3.input_layernorm.weight": "model-00002-of-00010.safetensors",
216
+ "model.layers.3.mlp.down_proj.weight": "model-00002-of-00010.safetensors",
217
+ "model.layers.3.mlp.gate_proj.weight": "model-00002-of-00010.safetensors",
218
+ "model.layers.3.mlp.up_proj.weight": "model-00002-of-00010.safetensors",
219
+ "model.layers.3.post_attention_layernorm.weight": "model-00002-of-00010.safetensors",
220
+ "model.layers.3.self_attn.k_proj.weight": "model-00001-of-00010.safetensors",
221
+ "model.layers.3.self_attn.o_proj.weight": "model-00001-of-00010.safetensors",
222
+ "model.layers.3.self_attn.q_proj.weight": "model-00001-of-00010.safetensors",
223
+ "model.layers.3.self_attn.v_proj.weight": "model-00001-of-00010.safetensors",
224
+ "model.layers.30.input_layernorm.weight": "model-00008-of-00010.safetensors",
225
+ "model.layers.30.mlp.down_proj.weight": "model-00008-of-00010.safetensors",
226
+ "model.layers.30.mlp.gate_proj.weight": "model-00008-of-00010.safetensors",
227
+ "model.layers.30.mlp.up_proj.weight": "model-00008-of-00010.safetensors",
228
+ "model.layers.30.post_attention_layernorm.weight": "model-00008-of-00010.safetensors",
229
+ "model.layers.30.self_attn.k_proj.weight": "model-00008-of-00010.safetensors",
230
+ "model.layers.30.self_attn.o_proj.weight": "model-00008-of-00010.safetensors",
231
+ "model.layers.30.self_attn.q_proj.weight": "model-00008-of-00010.safetensors",
232
+ "model.layers.30.self_attn.v_proj.weight": "model-00008-of-00010.safetensors",
233
+ "model.layers.31.input_layernorm.weight": "model-00008-of-00010.safetensors",
234
+ "model.layers.31.mlp.down_proj.weight": "model-00008-of-00010.safetensors",
235
+ "model.layers.31.mlp.gate_proj.weight": "model-00008-of-00010.safetensors",
236
+ "model.layers.31.mlp.up_proj.weight": "model-00008-of-00010.safetensors",
237
+ "model.layers.31.post_attention_layernorm.weight": "model-00008-of-00010.safetensors",
238
+ "model.layers.31.self_attn.k_proj.weight": "model-00008-of-00010.safetensors",
239
+ "model.layers.31.self_attn.o_proj.weight": "model-00008-of-00010.safetensors",
240
+ "model.layers.31.self_attn.q_proj.weight": "model-00008-of-00010.safetensors",
241
+ "model.layers.31.self_attn.v_proj.weight": "model-00008-of-00010.safetensors",
242
+ "model.layers.32.input_layernorm.weight": "model-00008-of-00010.safetensors",
243
+ "model.layers.32.mlp.down_proj.weight": "model-00008-of-00010.safetensors",
244
+ "model.layers.32.mlp.gate_proj.weight": "model-00008-of-00010.safetensors",
245
+ "model.layers.32.mlp.up_proj.weight": "model-00008-of-00010.safetensors",
246
+ "model.layers.32.post_attention_layernorm.weight": "model-00008-of-00010.safetensors",
247
+ "model.layers.32.self_attn.k_proj.weight": "model-00008-of-00010.safetensors",
248
+ "model.layers.32.self_attn.o_proj.weight": "model-00008-of-00010.safetensors",
249
+ "model.layers.32.self_attn.q_proj.weight": "model-00008-of-00010.safetensors",
250
+ "model.layers.32.self_attn.v_proj.weight": "model-00008-of-00010.safetensors",
251
+ "model.layers.33.input_layernorm.weight": "model-00009-of-00010.safetensors",
252
+ "model.layers.33.mlp.down_proj.weight": "model-00009-of-00010.safetensors",
253
+ "model.layers.33.mlp.gate_proj.weight": "model-00008-of-00010.safetensors",
254
+ "model.layers.33.mlp.up_proj.weight": "model-00009-of-00010.safetensors",
255
+ "model.layers.33.post_attention_layernorm.weight": "model-00009-of-00010.safetensors",
256
+ "model.layers.33.self_attn.k_proj.weight": "model-00008-of-00010.safetensors",
257
+ "model.layers.33.self_attn.o_proj.weight": "model-00008-of-00010.safetensors",
258
+ "model.layers.33.self_attn.q_proj.weight": "model-00008-of-00010.safetensors",
259
+ "model.layers.33.self_attn.v_proj.weight": "model-00008-of-00010.safetensors",
260
+ "model.layers.34.input_layernorm.weight": "model-00009-of-00010.safetensors",
261
+ "model.layers.34.mlp.down_proj.weight": "model-00009-of-00010.safetensors",
262
+ "model.layers.34.mlp.gate_proj.weight": "model-00009-of-00010.safetensors",
263
+ "model.layers.34.mlp.up_proj.weight": "model-00009-of-00010.safetensors",
264
+ "model.layers.34.post_attention_layernorm.weight": "model-00009-of-00010.safetensors",
265
+ "model.layers.34.self_attn.k_proj.weight": "model-00009-of-00010.safetensors",
266
+ "model.layers.34.self_attn.o_proj.weight": "model-00009-of-00010.safetensors",
267
+ "model.layers.34.self_attn.q_proj.weight": "model-00009-of-00010.safetensors",
268
+ "model.layers.34.self_attn.v_proj.weight": "model-00009-of-00010.safetensors",
269
+ "model.layers.35.input_layernorm.weight": "model-00009-of-00010.safetensors",
270
+ "model.layers.35.mlp.down_proj.weight": "model-00009-of-00010.safetensors",
271
+ "model.layers.35.mlp.gate_proj.weight": "model-00009-of-00010.safetensors",
272
+ "model.layers.35.mlp.up_proj.weight": "model-00009-of-00010.safetensors",
273
+ "model.layers.35.post_attention_layernorm.weight": "model-00009-of-00010.safetensors",
274
+ "model.layers.35.self_attn.k_proj.weight": "model-00009-of-00010.safetensors",
275
+ "model.layers.35.self_attn.o_proj.weight": "model-00009-of-00010.safetensors",
276
+ "model.layers.35.self_attn.q_proj.weight": "model-00009-of-00010.safetensors",
277
+ "model.layers.35.self_attn.v_proj.weight": "model-00009-of-00010.safetensors",
278
+ "model.layers.36.input_layernorm.weight": "model-00009-of-00010.safetensors",
279
+ "model.layers.36.mlp.down_proj.weight": "model-00009-of-00010.safetensors",
280
+ "model.layers.36.mlp.gate_proj.weight": "model-00009-of-00010.safetensors",
281
+ "model.layers.36.mlp.up_proj.weight": "model-00009-of-00010.safetensors",
282
+ "model.layers.36.post_attention_layernorm.weight": "model-00009-of-00010.safetensors",
283
+ "model.layers.36.self_attn.k_proj.weight": "model-00009-of-00010.safetensors",
284
+ "model.layers.36.self_attn.o_proj.weight": "model-00009-of-00010.safetensors",
285
+ "model.layers.36.self_attn.q_proj.weight": "model-00009-of-00010.safetensors",
286
+ "model.layers.36.self_attn.v_proj.weight": "model-00009-of-00010.safetensors",
287
+ "model.layers.37.input_layernorm.weight": "model-00010-of-00010.safetensors",
288
+ "model.layers.37.mlp.down_proj.weight": "model-00010-of-00010.safetensors",
289
+ "model.layers.37.mlp.gate_proj.weight": "model-00009-of-00010.safetensors",
290
+ "model.layers.37.mlp.up_proj.weight": "model-00009-of-00010.safetensors",
291
+ "model.layers.37.post_attention_layernorm.weight": "model-00010-of-00010.safetensors",
292
+ "model.layers.37.self_attn.k_proj.weight": "model-00009-of-00010.safetensors",
293
+ "model.layers.37.self_attn.o_proj.weight": "model-00009-of-00010.safetensors",
294
+ "model.layers.37.self_attn.q_proj.weight": "model-00009-of-00010.safetensors",
295
+ "model.layers.37.self_attn.v_proj.weight": "model-00009-of-00010.safetensors",
296
+ "model.layers.38.input_layernorm.weight": "model-00010-of-00010.safetensors",
297
+ "model.layers.38.mlp.down_proj.weight": "model-00010-of-00010.safetensors",
298
+ "model.layers.38.mlp.gate_proj.weight": "model-00010-of-00010.safetensors",
299
+ "model.layers.38.mlp.up_proj.weight": "model-00010-of-00010.safetensors",
300
+ "model.layers.38.post_attention_layernorm.weight": "model-00010-of-00010.safetensors",
301
+ "model.layers.38.self_attn.k_proj.weight": "model-00010-of-00010.safetensors",
302
+ "model.layers.38.self_attn.o_proj.weight": "model-00010-of-00010.safetensors",
303
+ "model.layers.38.self_attn.q_proj.weight": "model-00010-of-00010.safetensors",
304
+ "model.layers.38.self_attn.v_proj.weight": "model-00010-of-00010.safetensors",
305
+ "model.layers.39.input_layernorm.weight": "model-00010-of-00010.safetensors",
306
+ "model.layers.39.mlp.down_proj.weight": "model-00010-of-00010.safetensors",
307
+ "model.layers.39.mlp.gate_proj.weight": "model-00010-of-00010.safetensors",
308
+ "model.layers.39.mlp.up_proj.weight": "model-00010-of-00010.safetensors",
309
+ "model.layers.39.post_attention_layernorm.weight": "model-00010-of-00010.safetensors",
310
+ "model.layers.39.self_attn.k_proj.weight": "model-00010-of-00010.safetensors",
311
+ "model.layers.39.self_attn.o_proj.weight": "model-00010-of-00010.safetensors",
312
+ "model.layers.39.self_attn.q_proj.weight": "model-00010-of-00010.safetensors",
313
+ "model.layers.39.self_attn.v_proj.weight": "model-00010-of-00010.safetensors",
314
+ "model.layers.4.input_layernorm.weight": "model-00002-of-00010.safetensors",
315
+ "model.layers.4.mlp.down_proj.weight": "model-00002-of-00010.safetensors",
316
+ "model.layers.4.mlp.gate_proj.weight": "model-00002-of-00010.safetensors",
317
+ "model.layers.4.mlp.up_proj.weight": "model-00002-of-00010.safetensors",
318
+ "model.layers.4.post_attention_layernorm.weight": "model-00002-of-00010.safetensors",
319
+ "model.layers.4.self_attn.k_proj.weight": "model-00002-of-00010.safetensors",
320
+ "model.layers.4.self_attn.o_proj.weight": "model-00002-of-00010.safetensors",
321
+ "model.layers.4.self_attn.q_proj.weight": "model-00002-of-00010.safetensors",
322
+ "model.layers.4.self_attn.v_proj.weight": "model-00002-of-00010.safetensors",
323
+ "model.layers.5.input_layernorm.weight": "model-00002-of-00010.safetensors",
324
+ "model.layers.5.mlp.down_proj.weight": "model-00002-of-00010.safetensors",
325
+ "model.layers.5.mlp.gate_proj.weight": "model-00002-of-00010.safetensors",
326
+ "model.layers.5.mlp.up_proj.weight": "model-00002-of-00010.safetensors",
327
+ "model.layers.5.post_attention_layernorm.weight": "model-00002-of-00010.safetensors",
328
+ "model.layers.5.self_attn.k_proj.weight": "model-00002-of-00010.safetensors",
329
+ "model.layers.5.self_attn.o_proj.weight": "model-00002-of-00010.safetensors",
330
+ "model.layers.5.self_attn.q_proj.weight": "model-00002-of-00010.safetensors",
331
+ "model.layers.5.self_attn.v_proj.weight": "model-00002-of-00010.safetensors",
332
+ "model.layers.6.input_layernorm.weight": "model-00002-of-00010.safetensors",
333
+ "model.layers.6.mlp.down_proj.weight": "model-00002-of-00010.safetensors",
334
+ "model.layers.6.mlp.gate_proj.weight": "model-00002-of-00010.safetensors",
335
+ "model.layers.6.mlp.up_proj.weight": "model-00002-of-00010.safetensors",
336
+ "model.layers.6.post_attention_layernorm.weight": "model-00002-of-00010.safetensors",
337
+ "model.layers.6.self_attn.k_proj.weight": "model-00002-of-00010.safetensors",
338
+ "model.layers.6.self_attn.o_proj.weight": "model-00002-of-00010.safetensors",
339
+ "model.layers.6.self_attn.q_proj.weight": "model-00002-of-00010.safetensors",
340
+ "model.layers.6.self_attn.v_proj.weight": "model-00002-of-00010.safetensors",
341
+ "model.layers.7.input_layernorm.weight": "model-00003-of-00010.safetensors",
342
+ "model.layers.7.mlp.down_proj.weight": "model-00003-of-00010.safetensors",
343
+ "model.layers.7.mlp.gate_proj.weight": "model-00002-of-00010.safetensors",
344
+ "model.layers.7.mlp.up_proj.weight": "model-00003-of-00010.safetensors",
345
+ "model.layers.7.post_attention_layernorm.weight": "model-00003-of-00010.safetensors",
346
+ "model.layers.7.self_attn.k_proj.weight": "model-00002-of-00010.safetensors",
347
+ "model.layers.7.self_attn.o_proj.weight": "model-00002-of-00010.safetensors",
348
+ "model.layers.7.self_attn.q_proj.weight": "model-00002-of-00010.safetensors",
349
+ "model.layers.7.self_attn.v_proj.weight": "model-00002-of-00010.safetensors",
350
+ "model.layers.8.input_layernorm.weight": "model-00003-of-00010.safetensors",
351
+ "model.layers.8.mlp.down_proj.weight": "model-00003-of-00010.safetensors",
352
+ "model.layers.8.mlp.gate_proj.weight": "model-00003-of-00010.safetensors",
353
+ "model.layers.8.mlp.up_proj.weight": "model-00003-of-00010.safetensors",
354
+ "model.layers.8.post_attention_layernorm.weight": "model-00003-of-00010.safetensors",
355
+ "model.layers.8.self_attn.k_proj.weight": "model-00003-of-00010.safetensors",
356
+ "model.layers.8.self_attn.o_proj.weight": "model-00003-of-00010.safetensors",
357
+ "model.layers.8.self_attn.q_proj.weight": "model-00003-of-00010.safetensors",
358
+ "model.layers.8.self_attn.v_proj.weight": "model-00003-of-00010.safetensors",
359
+ "model.layers.9.input_layernorm.weight": "model-00003-of-00010.safetensors",
360
+ "model.layers.9.mlp.down_proj.weight": "model-00003-of-00010.safetensors",
361
+ "model.layers.9.mlp.gate_proj.weight": "model-00003-of-00010.safetensors",
362
+ "model.layers.9.mlp.up_proj.weight": "model-00003-of-00010.safetensors",
363
+ "model.layers.9.post_attention_layernorm.weight": "model-00003-of-00010.safetensors",
364
+ "model.layers.9.self_attn.k_proj.weight": "model-00003-of-00010.safetensors",
365
+ "model.layers.9.self_attn.o_proj.weight": "model-00003-of-00010.safetensors",
366
+ "model.layers.9.self_attn.q_proj.weight": "model-00003-of-00010.safetensors",
367
+ "model.layers.9.self_attn.v_proj.weight": "model-00003-of-00010.safetensors",
368
+ "model.norm.weight": "model-00010-of-00010.safetensors"
369
+ }
370
+ }
output-00001-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:886e017338edc90c42580300dbeea61df98879f75934dce2a93eb6ef8c50df77
3
+ size 8469171754
output-00002-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0c33e7ea48e57aced7fb55a2f19e1cefbb708f8a725340264070b5fc1c74cf35
3
+ size 8559229688
output-00003-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fa481564dade941f80140edbc23c167dc4f700ff0a5379d7eef2913921b8c116
3
+ size 1653908690
special_tokens_map.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": "<s>",
3
+ "eos_token": "<|im_end|>",
4
+ "pad_token": "<pad>",
5
+ "unk_token": "<unk>"
6
+ }
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e1e5daf1e619597fd56ff036925ea189916417c665d5bdd61b80199cf8d4e6d2
3
+ size 17078029
tokenizer_config.json ADDED
The diff for this file is too large to render. See raw diff