ritabratamaiti commited on
Commit
2258e59
1 Parent(s): 503036c

Upload tokenizer

Browse files
Files changed (2) hide show
  1. special_tokens_map.json +1 -0
  2. tokenizer_config.json +5 -13
special_tokens_map.json CHANGED
@@ -13,6 +13,7 @@
13
  "rstrip": false,
14
  "single_word": false
15
  },
 
16
  "unk_token": {
17
  "content": "<unk>",
18
  "lstrip": false,
 
13
  "rstrip": false,
14
  "single_word": false
15
  },
16
+ "pad_token": "<unk>",
17
  "unk_token": {
18
  "content": "<unk>",
19
  "lstrip": false,
tokenizer_config.json CHANGED
@@ -2,14 +2,6 @@
2
  "add_bos_token": true,
3
  "add_eos_token": false,
4
  "bos_token": {
5
- "__type": "AddedToken",
6
- "content": "",
7
- "lstrip": false,
8
- "normalized": true,
9
- "rstrip": false,
10
- "single_word": false
11
- },
12
- "clean_up_tokenization_spaces": {
13
  "__type": "AddedToken",
14
  "content": "<s>",
15
  "lstrip": false,
@@ -17,6 +9,7 @@
17
  "rstrip": false,
18
  "single_word": false
19
  },
 
20
  "eos_token": {
21
  "__type": "AddedToken",
22
  "content": "</s>",
@@ -25,10 +18,10 @@
25
  "rstrip": false,
26
  "single_word": false
27
  },
28
- "legacy": null,
29
- "model_max_length": 1000000000000000019884624838656,
30
  "pad_token": null,
31
- "padding_side": "right",
32
  "sp_model_kwargs": {},
33
  "tokenizer_class": "LlamaTokenizer",
34
  "unk_token": {
@@ -38,6 +31,5 @@
38
  "normalized": true,
39
  "rstrip": false,
40
  "single_word": false
41
- },
42
- "use_fast": false
43
  }
 
2
  "add_bos_token": true,
3
  "add_eos_token": false,
4
  "bos_token": {
 
 
 
 
 
 
 
 
5
  "__type": "AddedToken",
6
  "content": "<s>",
7
  "lstrip": false,
 
9
  "rstrip": false,
10
  "single_word": false
11
  },
12
+ "clean_up_tokenization_spaces": false,
13
  "eos_token": {
14
  "__type": "AddedToken",
15
  "content": "</s>",
 
18
  "rstrip": false,
19
  "single_word": false
20
  },
21
+ "legacy": true,
22
+ "model_max_length": 2048,
23
  "pad_token": null,
24
+ "padding_side": "left",
25
  "sp_model_kwargs": {},
26
  "tokenizer_class": "LlamaTokenizer",
27
  "unk_token": {
 
31
  "normalized": true,
32
  "rstrip": false,
33
  "single_word": false
34
+ }
 
35
  }