Abhaykoul commited on
Commit
7ebd09f
1 Parent(s): 45e4784

Upload 3 files

Browse files
Files changed (2) hide show
  1. special_tokens_map.json +6 -22
  2. tokenizer_config.json +178 -29
special_tokens_map.json CHANGED
@@ -1,38 +1,22 @@
1
  {
2
- "additional_special_tokens": [
3
- {
4
- "content": "<|im_start|>",
5
- "lstrip": false,
6
- "normalized": false,
7
- "rstrip": false,
8
- "single_word": false
9
- }
10
- ],
11
  "bos_token": {
12
- "content": "<|startoftext|>",
13
  "lstrip": false,
14
- "normalized": false,
15
  "rstrip": false,
16
  "single_word": false
17
  },
18
  "eos_token": {
19
- "content": "<|im_end|>",
20
  "lstrip": false,
21
- "normalized": false,
22
  "rstrip": false,
23
  "single_word": false
24
  },
25
  "pad_token": {
26
- "content": "<unk>",
27
  "lstrip": false,
28
- "normalized": false,
29
- "rstrip": false,
30
- "single_word": false
31
- },
32
- "unk_token": {
33
- "content": "<unk>",
34
- "lstrip": false,
35
- "normalized": false,
36
  "rstrip": false,
37
  "single_word": false
38
  }
 
1
  {
 
 
 
 
 
 
 
 
 
2
  "bos_token": {
3
+ "content": "<|begin▁of▁sentence|>",
4
  "lstrip": false,
5
+ "normalized": true,
6
  "rstrip": false,
7
  "single_word": false
8
  },
9
  "eos_token": {
10
+ "content": "<|EOT|>",
11
  "lstrip": false,
12
+ "normalized": true,
13
  "rstrip": false,
14
  "single_word": false
15
  },
16
  "pad_token": {
17
+ "content": "<|end▁of▁sentence|>",
18
  "lstrip": false,
19
+ "normalized": true,
 
 
 
 
 
 
 
20
  "rstrip": false,
21
  "single_word": false
22
  }
tokenizer_config.json CHANGED
@@ -1,62 +1,211 @@
1
  {
2
- "add_bos_token": false,
3
  "add_eos_token": false,
4
  "added_tokens_decoder": {
5
- "0": {
6
- "content": "<unk>",
7
  "lstrip": false,
8
- "normalized": false,
9
  "rstrip": false,
10
  "single_word": false,
11
- "special": true
12
  },
13
- "1": {
14
- "content": "<|startoftext|>",
15
  "lstrip": false,
16
- "normalized": false,
17
  "rstrip": false,
18
  "single_word": false,
19
- "special": true
20
  },
21
- "2": {
22
- "content": "<|endoftext|>",
23
  "lstrip": false,
24
- "normalized": false,
25
  "rstrip": false,
26
  "single_word": false,
27
- "special": true
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
28
  },
29
- "6": {
30
- "content": "<|im_start|>",
31
  "lstrip": false,
32
- "normalized": false,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
33
  "rstrip": false,
34
  "single_word": false,
35
  "special": true
36
  },
37
- "7": {
38
- "content": "<|im_end|>",
39
  "lstrip": false,
40
- "normalized": false,
41
  "rstrip": false,
42
  "single_word": false,
43
  "special": true
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
44
  }
45
  },
46
- "additional_special_tokens": [
47
- "<|im_start|>"
48
- ],
49
- "bos_token": "<|startoftext|>",
50
- "chat_template": "{% if messages[0]['role'] == 'system' %}{% set system_message = messages[0]['content'] %}{% endif %}{% if system_message is defined %}{{ '<|im_start|>system\\n' + system_message + '<|im_end|>\\n' }}{% endif %}{% for message in messages %}{% set content = message['content'] %}{% if message['role'] == 'user' %}{{ '<|im_start|>user\\n' + content + '<|im_end|>\\n<|im_start|>assistant\\n' }}{% elif message['role'] == 'assistant' %}{{ content + '<|im_end|>' + '\\n' }}{% endif %}{% endfor %}",
51
  "clean_up_tokenization_spaces": false,
52
- "eos_token": "<|im_end|>",
53
  "legacy": true,
54
- "model_max_length": 4096,
55
- "pad_token": "<unk>",
56
  "padding_side": "right",
57
  "sp_model_kwargs": {},
58
  "split_special_tokens": false,
59
  "tokenizer_class": "LlamaTokenizer",
60
- "unk_token": "<unk>",
61
  "use_default_system_prompt": false
62
- }
 
1
  {
2
+ "add_bos_token": true,
3
  "add_eos_token": false,
4
  "added_tokens_decoder": {
5
+ "32000": {
6
+ "content": "õ",
7
  "lstrip": false,
8
+ "normalized": true,
9
  "rstrip": false,
10
  "single_word": false,
11
+ "special": false
12
  },
13
+ "32001": {
14
+ "content": "÷",
15
  "lstrip": false,
16
+ "normalized": true,
17
  "rstrip": false,
18
  "single_word": false,
19
+ "special": false
20
  },
21
+ "32002": {
22
+ "content": "Á",
23
  "lstrip": false,
24
+ "normalized": true,
25
  "rstrip": false,
26
  "single_word": false,
27
+ "special": false
28
+ },
29
+ "32003": {
30
+ "content": "ý",
31
+ "lstrip": false,
32
+ "normalized": true,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": false
36
+ },
37
+ "32004": {
38
+ "content": "À",
39
+ "lstrip": false,
40
+ "normalized": true,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": false
44
+ },
45
+ "32005": {
46
+ "content": "ÿ",
47
+ "lstrip": false,
48
+ "normalized": true,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": false
52
+ },
53
+ "32006": {
54
+ "content": "ø",
55
+ "lstrip": false,
56
+ "normalized": true,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": false
60
+ },
61
+ "32007": {
62
+ "content": "ú",
63
+ "lstrip": false,
64
+ "normalized": true,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": false
68
+ },
69
+ "32008": {
70
+ "content": "þ",
71
+ "lstrip": false,
72
+ "normalized": true,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": false
76
+ },
77
+ "32009": {
78
+ "content": "ü",
79
+ "lstrip": false,
80
+ "normalized": true,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": false
84
+ },
85
+ "32010": {
86
+ "content": "ù",
87
+ "lstrip": false,
88
+ "normalized": true,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": false
92
  },
93
+ "32011": {
94
+ "content": "ö",
95
  "lstrip": false,
96
+ "normalized": true,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": false
100
+ },
101
+ "32012": {
102
+ "content": "û",
103
+ "lstrip": false,
104
+ "normalized": true,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": false
108
+ },
109
+ "32013": {
110
+ "content": "<|begin▁of▁sentence|>",
111
+ "lstrip": false,
112
+ "normalized": true,
113
  "rstrip": false,
114
  "single_word": false,
115
  "special": true
116
  },
117
+ "32014": {
118
+ "content": "<|end▁of▁sentence|>",
119
  "lstrip": false,
120
+ "normalized": true,
121
  "rstrip": false,
122
  "single_word": false,
123
  "special": true
124
+ },
125
+ "32015": {
126
+ "content": "<|fim▁hole|>",
127
+ "lstrip": false,
128
+ "normalized": true,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "32016": {
134
+ "content": "<|fim▁begin|>",
135
+ "lstrip": false,
136
+ "normalized": true,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "32017": {
142
+ "content": "<|fim▁end|>",
143
+ "lstrip": false,
144
+ "normalized": true,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "32018": {
150
+ "content": "<pad>",
151
+ "lstrip": false,
152
+ "normalized": true,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "32019": {
158
+ "content": "<|User|>",
159
+ "lstrip": false,
160
+ "normalized": true,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "32020": {
166
+ "content": "<|Assistant|>",
167
+ "lstrip": false,
168
+ "normalized": true,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "32021": {
174
+ "content": "<|EOT|>",
175
+ "lstrip": false,
176
+ "normalized": true,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ },
181
+ "32022": {
182
+ "content": "<API_RUN_START>",
183
+ "lstrip": false,
184
+ "normalized": true,
185
+ "rstrip": false,
186
+ "single_word": false,
187
+ "special": false
188
+ },
189
+ "32023": {
190
+ "content": "<API_RUN_STOP>",
191
+ "lstrip": false,
192
+ "normalized": true,
193
+ "rstrip": false,
194
+ "single_word": false,
195
+ "special": false
196
  }
197
  },
198
+ "bos_token": "<|begin▁of▁sentence|>",
199
+ "chat_template": "{% if messages[0]['role'] == 'system' %}{% set system_message = messages[0]['content'] %}{% endif %}{% if system_message is defined %}{{ system_message + '\\n' }}{% endif %}{% for message in messages %}{% set content = message['content'] %}{% if message['role'] == 'user' %}{{ 'Human: ' + content + '\\nAssistant: ' }}{% elif message['role'] == 'assistant' %}{{ content + '<|end▁of▁sentence|>' + '\\n' }}{% endif %}{% endfor %}",
 
 
 
200
  "clean_up_tokenization_spaces": false,
201
+ "eos_token": "<|EOT|>",
202
  "legacy": true,
203
+ "model_max_length": 16384,
204
+ "pad_token": "<|end▁of▁sentence|>",
205
  "padding_side": "right",
206
  "sp_model_kwargs": {},
207
  "split_special_tokens": false,
208
  "tokenizer_class": "LlamaTokenizer",
209
+ "unk_token": null,
210
  "use_default_system_prompt": false
211
+ }