Yak-hbdx commited on
Commit
2cbb476
1 Parent(s): dc2d920

Upload tokenizer

Browse files
Files changed (3) hide show
  1. special_tokens_map.json +9 -0
  2. tokenizer_config.json +16 -0
  3. vocab.txt +25 -0
special_tokens_map.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "pad_token": {
3
+ "content": "pad",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ }
9
+ }
tokenizer_config.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "added_tokens_decoder": {
3
+ "0": {
4
+ "content": "pad",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false,
9
+ "special": true
10
+ }
11
+ },
12
+ "clean_up_tokenization_spaces": true,
13
+ "model_max_length": 29,
14
+ "pad_token": "pad",
15
+ "tokenizer_class": "RnaTokenizer"
16
+ }
vocab.txt ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ AA
2
+ AC
3
+ AG
4
+ AT
5
+ CA
6
+ CC
7
+ CG
8
+ CT
9
+ GA
10
+ GC
11
+ GG
12
+ GT
13
+ TA
14
+ TC
15
+ TG
16
+ TT
17
+ pad
18
+ ((
19
+ (.
20
+ )(
21
+ ))
22
+ ).
23
+ .(
24
+ .)
25
+ ..