jpizarrom commited on
Commit
37f6165
1 Parent(s): 7b00cf2

Upload processor

Browse files
added_tokens.json CHANGED
@@ -1,8 +1,3 @@
1
  {
2
- "[CLS]": 101,
3
- "[DEC]": 30522,
4
- "[MASK]": 103,
5
- "[PAD]": 0,
6
- "[SEP]": 102,
7
- "[UNK]": 100
8
  }
 
1
  {
2
+ "[DEC]": 30522
 
 
 
 
 
3
  }
special_tokens_map.json CHANGED
@@ -1,5 +1,11 @@
1
  {
2
- "bos_token": "[DEC]",
 
 
 
 
 
 
3
  "cls_token": "[CLS]",
4
  "mask_token": "[MASK]",
5
  "pad_token": "[PAD]",
 
1
  {
2
+ "bos_token": {
3
+ "content": "[DEC]",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
  "cls_token": "[CLS]",
10
  "mask_token": "[MASK]",
11
  "pad_token": "[PAD]",
tokenizer_config.json CHANGED
@@ -6,7 +6,7 @@
6
  "normalized": false,
7
  "rstrip": false,
8
  "single_word": false,
9
- "special": false
10
  },
11
  "100": {
12
  "content": "[UNK]",
@@ -14,7 +14,7 @@
14
  "normalized": false,
15
  "rstrip": false,
16
  "single_word": false,
17
- "special": false
18
  },
19
  "101": {
20
  "content": "[CLS]",
@@ -22,7 +22,7 @@
22
  "normalized": false,
23
  "rstrip": false,
24
  "single_word": false,
25
- "special": false
26
  },
27
  "102": {
28
  "content": "[SEP]",
@@ -30,7 +30,7 @@
30
  "normalized": false,
31
  "rstrip": false,
32
  "single_word": false,
33
- "special": false
34
  },
35
  "103": {
36
  "content": "[MASK]",
@@ -38,18 +38,18 @@
38
  "normalized": false,
39
  "rstrip": false,
40
  "single_word": false,
41
- "special": false
42
  },
43
  "30522": {
44
  "content": "[DEC]",
45
- "lstrip": true,
46
  "normalized": false,
47
- "rstrip": true,
48
  "single_word": false,
49
  "special": true
50
  }
51
  },
52
- "additional_special_tokens": [],
53
  "clean_up_tokenization_spaces": true,
54
  "cls_token": "[CLS]",
55
  "do_basic_tokenize": true,
@@ -63,7 +63,6 @@
63
  "strip_accents": null,
64
  "tokenize_chinese_chars": true,
65
  "tokenizer_class": "BertTokenizer",
66
- "tokenizer_file": null,
67
  "truncation_side": "right",
68
  "unk_token": "[UNK]"
69
  }
 
6
  "normalized": false,
7
  "rstrip": false,
8
  "single_word": false,
9
+ "special": true
10
  },
11
  "100": {
12
  "content": "[UNK]",
 
14
  "normalized": false,
15
  "rstrip": false,
16
  "single_word": false,
17
+ "special": true
18
  },
19
  "101": {
20
  "content": "[CLS]",
 
22
  "normalized": false,
23
  "rstrip": false,
24
  "single_word": false,
25
+ "special": true
26
  },
27
  "102": {
28
  "content": "[SEP]",
 
30
  "normalized": false,
31
  "rstrip": false,
32
  "single_word": false,
33
+ "special": true
34
  },
35
  "103": {
36
  "content": "[MASK]",
 
38
  "normalized": false,
39
  "rstrip": false,
40
  "single_word": false,
41
+ "special": true
42
  },
43
  "30522": {
44
  "content": "[DEC]",
45
+ "lstrip": false,
46
  "normalized": false,
47
+ "rstrip": false,
48
  "single_word": false,
49
  "special": true
50
  }
51
  },
52
+ "bos_token": "[DEC]",
53
  "clean_up_tokenization_spaces": true,
54
  "cls_token": "[CLS]",
55
  "do_basic_tokenize": true,
 
63
  "strip_accents": null,
64
  "tokenize_chinese_chars": true,
65
  "tokenizer_class": "BertTokenizer",
 
66
  "truncation_side": "right",
67
  "unk_token": "[UNK]"
68
  }