Upload tokenizer

Files changed (5) hide show

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

special_tokens_map.json CHANGED Viewed

@@ -1,4 +1,11 @@
 {
   "eos_token": {
     "content": "<|endoftext|>",
     "lstrip": false,
@@ -12,5 +19,12 @@
     "normalized": false,
     "rstrip": false,
     "single_word": false
   }
 }

 {
+  "bos_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
   "eos_token": {
     "content": "<|endoftext|>",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,
     "single_word": false
+  },
+  "unk_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
   }
 }

tokenizer.json CHANGED Viewed

@@ -222,32 +222,7 @@
       }
     ]
   },
-  "post_processor": {
-    "type": "TemplateProcessing",
-    "single": [
-      {
-        "Sequence": {
-          "id": "A",
-          "type_id": 0
-        }
-      }
-    ],
-    "pair": [
-      {
-        "Sequence": {
-          "id": "A",
-          "type_id": 0
-        }
-      },
-      {
-        "Sequence": {
-          "id": "B",
-          "type_id": 1
-        }
-      }
-    ],
-    "special_tokens": {}
-  },
   "decoder": {
     "type": "ByteLevel",
     "add_prefix_space": true,

       }
     ]
   },
+  "post_processor": null,
   "decoder": {
     "type": "ByteLevel",
     "add_prefix_space": true,

tokenizer_config.json CHANGED Viewed

@@ -1,6 +1,4 @@
 {
-  "add_bos_token": false,
-  "add_eos_token": false,
   "add_prefix_space": false,
   "added_tokens_decoder": {
     "100256": {
@@ -180,12 +178,12 @@
       "special": true
     }
   },
-  "bos_token": null,
   "clean_up_tokenization_spaces": false,
   "eos_token": "<|endoftext|>",
   "extra_special_tokens": {},
   "model_max_length": 1000000000000000019884624838656,
   "pad_token": "<|pad|>",
-  "tokenizer_class": "GPTNeoXTokenizer",
-  "unk_token": null
 }

 {
   "add_prefix_space": false,
   "added_tokens_decoder": {
     "100256": {
       "special": true
     }
   },
+  "bos_token": "<|endoftext|>",
   "clean_up_tokenization_spaces": false,
   "eos_token": "<|endoftext|>",
   "extra_special_tokens": {},
   "model_max_length": 1000000000000000019884624838656,
   "pad_token": "<|pad|>",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "<|endoftext|>"
 }

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff