Upload processor

Files changed (3) hide show

added_tokens.json CHANGED Viewed

@@ -1,8 +1,3 @@
 {
-  "[CLS]": 101,
-  "[DEC]": 30522,
-  "[MASK]": 103,
-  "[PAD]": 0,
-  "[SEP]": 102,
-  "[UNK]": 100
 }

 {
+  "[DEC]": 30522
 }

special_tokens_map.json CHANGED Viewed

@@ -1,5 +1,11 @@
 {
-  "bos_token": "[DEC]",
   "cls_token": "[CLS]",
   "mask_token": "[MASK]",
   "pad_token": "[PAD]",

 {
+  "bos_token": {
+    "content": "[DEC]",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
   "cls_token": "[CLS]",
   "mask_token": "[MASK]",
   "pad_token": "[PAD]",

tokenizer_config.json CHANGED Viewed

@@ -6,7 +6,7 @@
       "normalized": false,
       "rstrip": false,
       "single_word": false,
-      "special": false
     },
     "100": {
       "content": "[UNK]",
@@ -14,7 +14,7 @@
       "normalized": false,
       "rstrip": false,
       "single_word": false,
-      "special": false
     },
     "101": {
       "content": "[CLS]",
@@ -22,7 +22,7 @@
       "normalized": false,
       "rstrip": false,
       "single_word": false,
-      "special": false
     },
     "102": {
       "content": "[SEP]",
@@ -30,7 +30,7 @@
       "normalized": false,
       "rstrip": false,
       "single_word": false,
-      "special": false
     },
     "103": {
       "content": "[MASK]",
@@ -38,18 +38,18 @@
       "normalized": false,
       "rstrip": false,
       "single_word": false,
-      "special": false
     },
     "30522": {
       "content": "[DEC]",
-      "lstrip": true,
       "normalized": false,
-      "rstrip": true,
       "single_word": false,
       "special": true
     }
   },
-  "additional_special_tokens": [],
   "clean_up_tokenization_spaces": true,
   "cls_token": "[CLS]",
   "do_basic_tokenize": true,
@@ -63,7 +63,6 @@
   "strip_accents": null,
   "tokenize_chinese_chars": true,
   "tokenizer_class": "BertTokenizer",
-  "tokenizer_file": null,
   "truncation_side": "right",
   "unk_token": "[UNK]"
 }

       "normalized": false,
       "rstrip": false,
       "single_word": false,
+      "special": true
     },
     "100": {
       "content": "[UNK]",
       "normalized": false,
       "rstrip": false,
       "single_word": false,
+      "special": true
     },
     "101": {
       "content": "[CLS]",
       "normalized": false,
       "rstrip": false,
       "single_word": false,
+      "special": true
     },
     "102": {
       "content": "[SEP]",
       "normalized": false,
       "rstrip": false,
       "single_word": false,
+      "special": true
     },
     "103": {
       "content": "[MASK]",
       "normalized": false,
       "rstrip": false,
       "single_word": false,
+      "special": true
     },
     "30522": {
       "content": "[DEC]",
+      "lstrip": false,
       "normalized": false,
+      "rstrip": false,
       "single_word": false,
       "special": true
     }
   },
+  "bos_token": "[DEC]",
   "clean_up_tokenization_spaces": true,
   "cls_token": "[CLS]",
   "do_basic_tokenize": true,
   "strip_accents": null,
   "tokenize_chinese_chars": true,
   "tokenizer_class": "BertTokenizer",
   "truncation_side": "right",
   "unk_token": "[UNK]"
 }