first commit

Browse files

Files changed (7) hide show

added_tokens.json +3 -0
config.json +63 -0
pytorch_model.bin +3 -0
special_tokens_map.json +15 -0
spm.model +3 -0
tokenizer.json +0 -0
tokenizer_config.json +58 -0

added_tokens.json ADDED Viewed

	@@ -0,0 +1,3 @@

+{
+  "[MASK]": 128000
+}

config.json ADDED Viewed

	@@ -0,0 +1,63 @@

+{
+  "base_model": "microsoft/deberta-v3-base",
+  "model_type": "deberta-v2",
+  "config_path": null,
+  "fc_dropout": 0.2,
+  "id2label": {
+    "0": "Adult",
+    "1": "Arts_and_Entertainment",
+    "10": "Health",
+    "11": "Hobbies_and_Leisure",
+    "12": "Home_and_Garden",
+    "13": "Internet_and_Telecom",
+    "14": "Jobs_and_Education",
+    "15": "Law_and_Government",
+    "16": "News",
+    "17": "Online_Communities",
+    "18": "People_and_Society",
+    "19": "Pets_and_Animals",
+    "2": "Autos_and_Vehicles",
+    "20": "Real_Estate",
+    "21": "Science",
+    "22": "Sensitive_Subjects",
+    "23": "Shopping",
+    "24": "Sports",
+    "25": "Travel_and_Transportation",
+    "3": "Beauty_and_Fitness",
+    "4": "Books_and_Literature",
+    "5": "Business_and_Industrial",
+    "6": "Computers_and_Electronics",
+    "7": "Finance",
+    "8": "Food_and_Drink",
+    "9": "Games"
+  },
+  "label2id": {
+    "Adult": 0,
+    "Arts_and_Entertainment": 1,
+    "Autos_and_Vehicles": 2,
+    "Beauty_and_Fitness": 3,
+    "Books_and_Literature": 4,
+    "Business_and_Industrial": 5,
+    "Computers_and_Electronics": 6,
+    "Finance": 7,
+    "Food_and_Drink": 8,
+    "Games": 9,
+    "Health": 10,
+    "Hobbies_and_Leisure": 11,
+    "Home_and_Garden": 12,
+    "Internet_and_Telecom": 13,
+    "Jobs_and_Education": 14,
+    "Law_and_Government": 15,
+    "News": 16,
+    "Online_Communities": 17,
+    "People_and_Society": 18,
+    "Pets_and_Animals": 19,
+    "Real_Estate": 20,
+    "Science": 21,
+    "Sensitive_Subjects": 22,
+    "Shopping": 23,
+    "Sports": 24,
+    "Travel_and_Transportation": 25
+  },
+  "pretrained": true
+}

pytorch_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:96ad552a3ddc6228efcf45df7b42ad214547a9b7afb98548c6a65515c074ab94
+size 735487530

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,15 @@

+{
+  "bos_token": "[CLS]",
+  "cls_token": "[CLS]",
+  "eos_token": "[SEP]",
+  "mask_token": "[MASK]",
+  "pad_token": "[PAD]",
+  "sep_token": "[SEP]",
+  "unk_token": {
+    "content": "[UNK]",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  }
+}

spm.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c679fbf93643d19aab7ee10c0b99e460bdbc02fedf34b92b05af343b4af586fd
+size 2464616

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,58 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "[PAD]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "[CLS]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "[SEP]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "[UNK]",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "128000": {
+      "content": "[MASK]",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "[CLS]",
+  "clean_up_tokenization_spaces": true,
+  "cls_token": "[CLS]",
+  "do_lower_case": false,
+  "eos_token": "[SEP]",
+  "mask_token": "[MASK]",
+  "model_max_length": 512,
+  "pad_token": "[PAD]",
+  "sep_token": "[SEP]",
+  "sp_model_kwargs": {},
+  "split_by_punct": false,
+  "tokenizer_class": "DebertaV2Tokenizer",
+  "unk_token": "[UNK]",
+  "vocab_type": "spm"
+}