Abhaykoul commited on
Commit
69ce19d
1 Parent(s): d1771a6

Update tokenizer_config.json

Browse files
Files changed (1) hide show
  1. tokenizer_config.json +143 -78
tokenizer_config.json CHANGED
@@ -1,4 +1,6 @@
1
  {
 
 
2
  "add_prefix_space": false,
3
  "added_tokens_decoder": {
4
  "0": {
@@ -10,150 +12,206 @@
10
  "special": true
11
  },
12
  "1": {
13
- "content": "<fim_prefix>",
14
  "lstrip": false,
15
  "normalized": false,
16
  "rstrip": false,
17
  "single_word": false,
18
  "special": true
19
  },
20
- "2": {
21
- "content": "<fim_middle>",
22
  "lstrip": false,
23
- "normalized": false,
24
  "rstrip": false,
25
  "single_word": false,
26
- "special": true
27
  },
28
- "3": {
29
- "content": "<fim_suffix>",
30
  "lstrip": false,
31
- "normalized": false,
32
  "rstrip": false,
33
  "single_word": false,
34
- "special": true
35
  },
36
- "4": {
37
- "content": "<fim_pad>",
38
  "lstrip": false,
39
- "normalized": false,
40
  "rstrip": false,
41
  "single_word": false,
42
- "special": true
43
  },
44
- "5": {
45
- "content": "<filename>",
46
  "lstrip": false,
47
- "normalized": false,
48
  "rstrip": false,
49
  "single_word": false,
50
- "special": true
51
  },
52
- "6": {
53
- "content": "<gh_stars>",
54
  "lstrip": false,
55
- "normalized": false,
56
  "rstrip": false,
57
  "single_word": false,
58
- "special": true
59
  },
60
- "7": {
61
- "content": "<issue_start>",
62
  "lstrip": false,
63
- "normalized": false,
64
  "rstrip": false,
65
  "single_word": false,
66
- "special": true
67
  },
68
- "8": {
69
- "content": "<issue_comment>",
70
  "lstrip": false,
71
- "normalized": false,
72
  "rstrip": false,
73
  "single_word": false,
74
- "special": true
75
  },
76
- "9": {
77
- "content": "<issue_closed>",
78
  "lstrip": false,
79
- "normalized": false,
80
  "rstrip": false,
81
  "single_word": false,
82
- "special": true
83
  },
84
- "10": {
85
- "content": "<jupyter_start>",
86
  "lstrip": false,
87
- "normalized": false,
88
  "rstrip": false,
89
  "single_word": false,
90
- "special": true
91
  },
92
- "11": {
93
- "content": "<jupyter_text>",
94
  "lstrip": false,
95
- "normalized": false,
96
  "rstrip": false,
97
  "single_word": false,
98
- "special": true
99
  },
100
- "12": {
101
- "content": "<jupyter_code>",
102
  "lstrip": false,
103
- "normalized": false,
104
  "rstrip": false,
105
  "single_word": false,
106
- "special": true
107
  },
108
- "13": {
109
- "content": "<jupyter_output>",
110
  "lstrip": false,
111
- "normalized": false,
112
  "rstrip": false,
113
  "single_word": false,
114
- "special": true
115
  },
116
- "14": {
117
- "content": "<empty_output>",
118
  "lstrip": false,
119
- "normalized": false,
120
  "rstrip": false,
121
  "single_word": false,
122
- "special": true
123
  },
124
- "15": {
125
- "content": "<commit_before>",
126
  "lstrip": false,
127
- "normalized": false,
128
  "rstrip": false,
129
  "single_word": false,
130
- "special": true
131
  },
132
- "16": {
133
- "content": "<commit_msg>",
134
  "lstrip": false,
135
- "normalized": false,
136
  "rstrip": false,
137
  "single_word": false,
138
- "special": true
139
  },
140
- "17": {
141
- "content": "<commit_after>",
142
  "lstrip": false,
143
- "normalized": false,
144
  "rstrip": false,
145
  "single_word": false,
146
- "special": true
147
  },
148
- "18": {
149
- "content": "<reponame>",
150
  "lstrip": false,
151
- "normalized": false,
152
  "rstrip": false,
153
  "single_word": false,
154
- "special": true
155
  },
156
- "49152": {
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
157
  "content": "<|im_start|>",
158
  "lstrip": false,
159
  "normalized": false,
@@ -161,28 +219,35 @@
161
  "single_word": false,
162
  "special": true
163
  },
164
- "49153": {
165
  "content": "<|im_end|>",
166
  "lstrip": false,
167
  "normalized": false,
168
  "rstrip": false,
169
  "single_word": false,
170
  "special": true
 
 
 
 
 
 
 
 
171
  }
172
  },
173
  "additional_special_tokens": [
174
  "<|im_start|>",
175
  "<|im_end|>"
176
  ],
177
- "bos_token": "<|endoftext|>",
178
  "chat_template": "{% for message in messages %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
179
  "clean_up_tokenization_spaces": true,
180
- "eos_token": "<|endoftext|>",
181
- "model_max_length": 9223372036854775807,
182
- "pad_token": "<|endoftext|>",
183
- "padding_side": "left",
184
- "tokenizer_class": "GPT2Tokenizer",
185
- "unk_token": "<|endoftext|>",
186
  "vocab_size": 49152
187
  }
188
 
 
1
  {
2
+ "add_bos_token": false,
3
+ "add_eos_token": false,
4
  "add_prefix_space": false,
5
  "added_tokens_decoder": {
6
  "0": {
 
12
  "special": true
13
  },
14
  "1": {
15
+ "content": "<|padding|>",
16
  "lstrip": false,
17
  "normalized": false,
18
  "rstrip": false,
19
  "single_word": false,
20
  "special": true
21
  },
22
+ "50254": {
23
+ "content": " ",
24
  "lstrip": false,
25
+ "normalized": true,
26
  "rstrip": false,
27
  "single_word": false,
28
+ "special": false
29
  },
30
+ "50255": {
31
+ "content": " ",
32
  "lstrip": false,
33
+ "normalized": true,
34
  "rstrip": false,
35
  "single_word": false,
36
+ "special": false
37
  },
38
+ "50256": {
39
+ "content": " ",
40
  "lstrip": false,
41
+ "normalized": true,
42
  "rstrip": false,
43
  "single_word": false,
44
+ "special": false
45
  },
46
+ "50257": {
47
+ "content": " ",
48
  "lstrip": false,
49
+ "normalized": true,
50
  "rstrip": false,
51
  "single_word": false,
52
+ "special": false
53
  },
54
+ "50258": {
55
+ "content": " ",
56
  "lstrip": false,
57
+ "normalized": true,
58
  "rstrip": false,
59
  "single_word": false,
60
+ "special": false
61
  },
62
+ "50259": {
63
+ "content": " ",
64
  "lstrip": false,
65
+ "normalized": true,
66
  "rstrip": false,
67
  "single_word": false,
68
+ "special": false
69
  },
70
+ "50260": {
71
+ "content": " ",
72
  "lstrip": false,
73
+ "normalized": true,
74
  "rstrip": false,
75
  "single_word": false,
76
+ "special": false
77
  },
78
+ "50261": {
79
+ "content": " ",
80
  "lstrip": false,
81
+ "normalized": true,
82
  "rstrip": false,
83
  "single_word": false,
84
+ "special": false
85
  },
86
+ "50262": {
87
+ "content": " ",
88
  "lstrip": false,
89
+ "normalized": true,
90
  "rstrip": false,
91
  "single_word": false,
92
+ "special": false
93
  },
94
+ "50263": {
95
+ "content": " ",
96
  "lstrip": false,
97
+ "normalized": true,
98
  "rstrip": false,
99
  "single_word": false,
100
+ "special": false
101
  },
102
+ "50264": {
103
+ "content": " ",
104
  "lstrip": false,
105
+ "normalized": true,
106
  "rstrip": false,
107
  "single_word": false,
108
+ "special": false
109
  },
110
+ "50265": {
111
+ "content": " ",
112
  "lstrip": false,
113
+ "normalized": true,
114
  "rstrip": false,
115
  "single_word": false,
116
+ "special": false
117
  },
118
+ "50266": {
119
+ "content": " ",
120
  "lstrip": false,
121
+ "normalized": true,
122
  "rstrip": false,
123
  "single_word": false,
124
+ "special": false
125
  },
126
+ "50267": {
127
+ "content": " ",
128
  "lstrip": false,
129
+ "normalized": true,
130
  "rstrip": false,
131
  "single_word": false,
132
+ "special": false
133
  },
134
+ "50268": {
135
+ "content": " ",
136
  "lstrip": false,
137
+ "normalized": true,
138
  "rstrip": false,
139
  "single_word": false,
140
+ "special": false
141
  },
142
+ "50269": {
143
+ "content": " ",
144
  "lstrip": false,
145
+ "normalized": true,
146
  "rstrip": false,
147
  "single_word": false,
148
+ "special": false
149
  },
150
+ "50270": {
151
+ "content": " ",
152
  "lstrip": false,
153
+ "normalized": true,
154
  "rstrip": false,
155
  "single_word": false,
156
+ "special": false
157
  },
158
+ "50271": {
159
+ "content": " ",
160
+ "lstrip": false,
161
+ "normalized": true,
162
+ "rstrip": false,
163
+ "single_word": false,
164
+ "special": false
165
+ },
166
+ "50272": {
167
+ "content": " ",
168
+ "lstrip": false,
169
+ "normalized": true,
170
+ "rstrip": false,
171
+ "single_word": false,
172
+ "special": false
173
+ },
174
+ "50273": {
175
+ "content": " ",
176
+ "lstrip": false,
177
+ "normalized": true,
178
+ "rstrip": false,
179
+ "single_word": false,
180
+ "special": false
181
+ },
182
+ "50274": {
183
+ "content": " ",
184
+ "lstrip": false,
185
+ "normalized": true,
186
+ "rstrip": false,
187
+ "single_word": false,
188
+ "special": false
189
+ },
190
+ "50275": {
191
+ "content": " ",
192
+ "lstrip": false,
193
+ "normalized": true,
194
+ "rstrip": false,
195
+ "single_word": false,
196
+ "special": false
197
+ },
198
+ "50276": {
199
+ "content": " ",
200
+ "lstrip": false,
201
+ "normalized": true,
202
+ "rstrip": false,
203
+ "single_word": false,
204
+ "special": false
205
+ },
206
+ "50277": {
207
+ "content": "<|pad|>",
208
+ "lstrip": false,
209
+ "normalized": true,
210
+ "rstrip": false,
211
+ "single_word": false,
212
+ "special": false
213
+ },
214
+ "50278": {
215
  "content": "<|im_start|>",
216
  "lstrip": false,
217
  "normalized": false,
 
219
  "single_word": false,
220
  "special": true
221
  },
222
+ "50279": {
223
  "content": "<|im_end|>",
224
  "lstrip": false,
225
  "normalized": false,
226
  "rstrip": false,
227
  "single_word": false,
228
  "special": true
229
+ },
230
+ "50280": {
231
+ "content": "[PAD]",
232
+ "lstrip": false,
233
+ "normalized": false,
234
+ "rstrip": false,
235
+ "single_word": false,
236
+ "special": true
237
  }
238
  },
239
  "additional_special_tokens": [
240
  "<|im_start|>",
241
  "<|im_end|>"
242
  ],
243
+ "bos_token": "<|im_start|>",
244
  "chat_template": "{% for message in messages %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
245
  "clean_up_tokenization_spaces": true,
246
+ "eos_token": "<|im_end|>",
247
+ "model_max_length": 1000000000000000019884624838656,
248
+ "pad_token": "<|im_end|>",
249
+ "tokenizer_class": "GPTNeoXTokenizer",
250
+ "unk_token": "<|endoftext|>"
 
251
  "vocab_size": 49152
252
  }
253