Oztobuzz commited on
Commit
d00e5e7
1 Parent(s): 0068744

Upload tokenizer

Browse files
Files changed (2) hide show
  1. tokenizer.json +0 -0
  2. tokenizer_config.json +5 -5
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json CHANGED
@@ -9,7 +9,7 @@
9
  "special": true
10
  },
11
  "1": {
12
- "content": "<pad>",
13
  "lstrip": false,
14
  "normalized": false,
15
  "rstrip": false,
@@ -17,7 +17,7 @@
17
  "special": true
18
  },
19
  "2": {
20
- "content": "</s>",
21
  "lstrip": false,
22
  "normalized": false,
23
  "rstrip": false,
@@ -25,14 +25,14 @@
25
  "special": true
26
  },
27
  "3": {
28
- "content": "<unk>",
29
  "lstrip": false,
30
  "normalized": false,
31
  "rstrip": false,
32
  "single_word": false,
33
  "special": true
34
  },
35
- "64000": {
36
  "content": "<mask>",
37
  "lstrip": false,
38
  "normalized": false,
@@ -49,6 +49,6 @@
49
  "model_max_length": 1000000000000000019884624838656,
50
  "pad_token": "</s>",
51
  "sep_token": "</s>",
52
- "tokenizer_class": "PhobertTokenizer",
53
  "unk_token": "<unk>"
54
  }
 
9
  "special": true
10
  },
11
  "1": {
12
+ "content": "</s>",
13
  "lstrip": false,
14
  "normalized": false,
15
  "rstrip": false,
 
17
  "special": true
18
  },
19
  "2": {
20
+ "content": "<unk>",
21
  "lstrip": false,
22
  "normalized": false,
23
  "rstrip": false,
 
25
  "special": true
26
  },
27
  "3": {
28
+ "content": "<pad>",
29
  "lstrip": false,
30
  "normalized": false,
31
  "rstrip": false,
32
  "single_word": false,
33
  "special": true
34
  },
35
+ "4": {
36
  "content": "<mask>",
37
  "lstrip": false,
38
  "normalized": false,
 
49
  "model_max_length": 1000000000000000019884624838656,
50
  "pad_token": "</s>",
51
  "sep_token": "</s>",
52
+ "tokenizer_class": "PreTrainedTokenizerFast",
53
  "unk_token": "<unk>"
54
  }