File size: 602 Bytes
209ed55
 
665571c
209ed55
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
name: bpe_tokenizer
config_type: preprocessor
pretrained_path: hezarai/trocr-tiny-fa
max_length: 512
truncation_strategy: longest_first
truncation_direction: right
stride: 0
padding_strategy: longest
padding_direction: right
pad_to_multiple_of: 0
pad_token_id: 0
pad_token: <pad>
pad_token_type_id: 0
unk_token: <unk>
special_tokens:
- <s>
- <pad>
- </s>
- <unk>
- <mask>
- <|endoftext|>
- <|startoftext|>
- <nl>
- <hs>
- <sep>
- <cls>
continuing_subword_prefix: ''
end_of_word_suffix: ''
fuse_unk: false
vocab_size: 42000
min_frequency: 2
limit_alphabet: 1000
initial_alphabet: []
show_progress: true