nihalbaig commited on
Commit
3c09ed6
1 Parent(s): 83340ec

Upload lm-boosted decoder

Browse files
added_tokens.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"<s>": 78, "</s>": 79}
alphabet.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"labels": [" ", "=", "\u0964", "\u0981", "\u0982", "\u0983", "\u0985", "\u0986", "\u0987", "\u0988", "\u0989", "\u098a", "\u098b", "\u098f", "\u0990", "\u0993", "\u0994", "\u0995", "\u0996", "\u0997", "\u0998", "\u0999", "\u099a", "\u099b", "\u099c", "\u099d", "\u099e", "\u099f", "\u09a0", "\u09a1", "\u09a2", "\u09a3", "\u09a4", "\u09a5", "\u09a6", "\u09a7", "\u09a8", "\u09aa", "\u09ab", "\u09ac", "\u09ad", "\u09ae", "\u09af", "\u09b0", "\u09b2", "\u09b6", "\u09b7", "\u09b8", "\u09b9", "\u09be", "\u09bf", "\u09c0", "\u09c1", "\u09c2", "\u09c3", "\u09c7", "\u09c8", "\u09cb", "\u09cc", "\u09cd", "\u09ce", "\u09dc", "\u09dd", "\u09df", "\u09e6", "\u09e7", "\u09e8", "\u09e9", "\u09ea", "\u09eb", "\u09ec", "\u09ed", "\u09ee", "\u09ef", "\u200d", "\u2014", "\u2047", "", "<s>", "</s>"], "is_bpe": false}
language_model/5gram.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8c44052d422ad3a22ea7bb32c6582be9746544bfdf88cb2d15050773758b3047
3
+ size 111910037
language_model/attrs.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"alpha": 0.5, "beta": 1.5, "unk_score_offset": -10.0, "score_boundary": true}
language_model/unigrams.txt ADDED
The diff for this file is too large to render. See raw diff
 
preprocessor_config.json ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "do_normalize": true,
3
+ "feature_extractor_type": "Wav2Vec2FeatureExtractor",
4
+ "feature_size": 1,
5
+ "padding_side": "right",
6
+ "padding_value": 0.0,
7
+ "processor_class": "Wav2Vec2ProcessorWithLM",
8
+ "return_attention_mask": true,
9
+ "sampling_rate": 16000
10
+ }
special_tokens_map.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"bos_token": "<s>", "eos_token": "</s>", "unk_token": "[UNK]", "pad_token": "[PAD]", "additional_special_tokens": [{"content": "<s>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true}, {"content": "</s>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true}]}
tokenizer_config.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"unk_token": "[UNK]", "bos_token": "<s>", "eos_token": "</s>", "pad_token": "[PAD]", "do_lower_case": false, "word_delimiter_token": "|", "replace_word_delimiter_char": " ", "processor_class": "Wav2Vec2ProcessorWithLM", "special_tokens_map_file": "/root/.cache/huggingface/transformers/0a23835ad81968ddc87c2c1f6316ef42ece397bd53fe85c6d657e1fbc40e163e.a21d51735cf8667bcd610f057e88548d5d6a381401f6b4501a8bc6c1a9dc8498", "name_or_path": "nihalbaig/wav2vec2-large-xlsr-bn", "tokenizer_class": "Wav2Vec2CTCTokenizer"}
vocab.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"=": 1, "।": 2, "ঁ": 3, "ং": 4, "ঃ": 5, "অ": 6, "আ": 7, "ই": 8, "ঈ": 9, "উ": 10, "ঊ": 11, "ঋ": 12, "এ": 13, "ঐ": 14, "ও": 15, "ঔ": 16, "ক": 17, "খ": 18, "গ": 19, "ঘ": 20, "ঙ": 21, "চ": 22, "ছ": 23, "জ": 24, "ঝ": 25, "ঞ": 26, "ট": 27, "ঠ": 28, "ড": 29, "ঢ": 30, "ণ": 31, "ত": 32, "থ": 33, "দ": 34, "ধ": 35, "ন": 36, "প": 37, "ফ": 38, "ব": 39, "ভ": 40, "ম": 41, "য": 42, "র": 43, "ল": 44, "শ": 45, "ষ": 46, "স": 47, "হ": 48, "া": 49, "ি": 50, "ী": 51, "ু": 52, "ূ": 53, "ৃ": 54, "ে": 55, "ৈ": 56, "ো": 57, "ৌ": 58, "্": 59, "ৎ": 60, "ড়": 61, "ঢ়": 62, "য়": 63, "০": 64, "১": 65, "২": 66, "৩": 67, "৪": 68, "৫": 69, "৬": 70, "৭": 71, "৮": 72, "৯": 73, "‍": 74, "—": 75, "|": 0, "[UNK]": 76, "[PAD]": 77}