xchen16 commited on
Commit
b5fb439
·
verified ·
1 Parent(s): b74589a

Upload tokenizer

Browse files
added_tokens.json CHANGED
@@ -1,5 +1,7 @@
1
  {
 
2
  "[MASK]": 50,
3
  "[PAD]": 49,
 
4
  "[UNK]": 48
5
  }
 
1
  {
2
+ "[CLS]": 52,
3
  "[MASK]": 50,
4
  "[PAD]": 49,
5
+ "[SEP]": 51,
6
  "[UNK]": 48
7
  }
special_tokens_map.json CHANGED
@@ -1,4 +1,5 @@
1
  {
 
2
  "mask_token": {
3
  "content": "[MASK]",
4
  "lstrip": false,
@@ -13,6 +14,7 @@
13
  "rstrip": false,
14
  "single_word": false
15
  },
 
16
  "unk_token": {
17
  "content": "[UNK]",
18
  "lstrip": false,
 
1
  {
2
+ "cls_token": "[CLS]",
3
  "mask_token": {
4
  "content": "[MASK]",
5
  "lstrip": false,
 
14
  "rstrip": false,
15
  "single_word": false
16
  },
17
+ "sep_token": "[SEP]",
18
  "unk_token": {
19
  "content": "[UNK]",
20
  "lstrip": false,
tokenizer.json CHANGED
@@ -29,6 +29,24 @@
29
  "rstrip": false,
30
  "normalized": false,
31
  "special": true
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
32
  }
33
  ],
34
  "normalizer": {
 
29
  "rstrip": false,
30
  "normalized": false,
31
  "special": true
32
+ },
33
+ {
34
+ "id": 51,
35
+ "content": "[SEP]",
36
+ "single_word": false,
37
+ "lstrip": false,
38
+ "rstrip": false,
39
+ "normalized": false,
40
+ "special": true
41
+ },
42
+ {
43
+ "id": 52,
44
+ "content": "[CLS]",
45
+ "single_word": false,
46
+ "lstrip": false,
47
+ "rstrip": false,
48
+ "normalized": false,
49
+ "special": true
50
  }
51
  ],
52
  "normalizer": {
tokenizer_config.json CHANGED
@@ -23,15 +23,34 @@
23
  "rstrip": false,
24
  "single_word": false,
25
  "special": true
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
26
  }
27
  },
28
  "clean_up_tokenization_spaces": false,
 
29
  "do_lower_case": false,
30
  "extra_special_tokens": {},
31
  "mask_token": "[MASK]",
32
  "model_max_length": 207,
33
  "pad_token": "[PAD]",
 
34
  "strip_accents": null,
35
- "tokenizer_class": "G2PTTokenizer",
 
36
  "unk_token": "[UNK]"
37
  }
 
23
  "rstrip": false,
24
  "single_word": false,
25
  "special": true
26
+ },
27
+ "51": {
28
+ "content": "[SEP]",
29
+ "lstrip": false,
30
+ "normalized": false,
31
+ "rstrip": false,
32
+ "single_word": false,
33
+ "special": true
34
+ },
35
+ "52": {
36
+ "content": "[CLS]",
37
+ "lstrip": false,
38
+ "normalized": false,
39
+ "rstrip": false,
40
+ "single_word": false,
41
+ "special": true
42
  }
43
  },
44
  "clean_up_tokenization_spaces": false,
45
+ "cls_token": "[CLS]",
46
  "do_lower_case": false,
47
  "extra_special_tokens": {},
48
  "mask_token": "[MASK]",
49
  "model_max_length": 207,
50
  "pad_token": "[PAD]",
51
+ "sep_token": "[SEP]",
52
  "strip_accents": null,
53
+ "tokenize_chinese_chars": true,
54
+ "tokenizer_class": "BertTokenizer",
55
  "unk_token": "[UNK]"
56
  }