Upload lm-boosted decoder
Browse files- added_tokens.json +1 -0
- alphabet.json +1 -0
- language_model/5gram.bin +3 -0
- language_model/attrs.json +1 -0
- language_model/unigrams.txt +0 -0
- preprocessor_config.json +10 -0
- special_tokens_map.json +1 -0
- tokenizer_config.json +1 -0
- vocab.json +1 -0
added_tokens.json
ADDED
@@ -0,0 +1 @@
|
|
|
|
|
1 |
+
{"<s>": 78, "</s>": 79}
|
alphabet.json
ADDED
@@ -0,0 +1 @@
|
|
|
|
|
1 |
+
{"labels": [" ", "=", "\u0964", "\u0981", "\u0982", "\u0983", "\u0985", "\u0986", "\u0987", "\u0988", "\u0989", "\u098a", "\u098b", "\u098f", "\u0990", "\u0993", "\u0994", "\u0995", "\u0996", "\u0997", "\u0998", "\u0999", "\u099a", "\u099b", "\u099c", "\u099d", "\u099e", "\u099f", "\u09a0", "\u09a1", "\u09a2", "\u09a3", "\u09a4", "\u09a5", "\u09a6", "\u09a7", "\u09a8", "\u09aa", "\u09ab", "\u09ac", "\u09ad", "\u09ae", "\u09af", "\u09b0", "\u09b2", "\u09b6", "\u09b7", "\u09b8", "\u09b9", "\u09be", "\u09bf", "\u09c0", "\u09c1", "\u09c2", "\u09c3", "\u09c7", "\u09c8", "\u09cb", "\u09cc", "\u09cd", "\u09ce", "\u09dc", "\u09dd", "\u09df", "\u09e6", "\u09e7", "\u09e8", "\u09e9", "\u09ea", "\u09eb", "\u09ec", "\u09ed", "\u09ee", "\u09ef", "\u200d", "\u2014", "\u2047", "", "<s>", "</s>"], "is_bpe": false}
|
language_model/5gram.bin
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:8c44052d422ad3a22ea7bb32c6582be9746544bfdf88cb2d15050773758b3047
|
3 |
+
size 111910037
|
language_model/attrs.json
ADDED
@@ -0,0 +1 @@
|
|
|
|
|
1 |
+
{"alpha": 0.5, "beta": 1.5, "unk_score_offset": -10.0, "score_boundary": true}
|
language_model/unigrams.txt
ADDED
The diff for this file is too large to render.
See raw diff
|
|
preprocessor_config.json
ADDED
@@ -0,0 +1,10 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
1 |
+
{
|
2 |
+
"do_normalize": true,
|
3 |
+
"feature_extractor_type": "Wav2Vec2FeatureExtractor",
|
4 |
+
"feature_size": 1,
|
5 |
+
"padding_side": "right",
|
6 |
+
"padding_value": 0.0,
|
7 |
+
"processor_class": "Wav2Vec2ProcessorWithLM",
|
8 |
+
"return_attention_mask": true,
|
9 |
+
"sampling_rate": 16000
|
10 |
+
}
|
special_tokens_map.json
ADDED
@@ -0,0 +1 @@
|
|
|
|
|
1 |
+
{"bos_token": "<s>", "eos_token": "</s>", "unk_token": "[UNK]", "pad_token": "[PAD]", "additional_special_tokens": [{"content": "<s>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true}, {"content": "</s>", "single_word": false, "lstrip": false, "rstrip": false, "normalized": true}]}
|
tokenizer_config.json
ADDED
@@ -0,0 +1 @@
|
|
|
|
|
1 |
+
{"unk_token": "[UNK]", "bos_token": "<s>", "eos_token": "</s>", "pad_token": "[PAD]", "do_lower_case": false, "word_delimiter_token": "|", "replace_word_delimiter_char": " ", "processor_class": "Wav2Vec2ProcessorWithLM", "special_tokens_map_file": "/root/.cache/huggingface/transformers/0a23835ad81968ddc87c2c1f6316ef42ece397bd53fe85c6d657e1fbc40e163e.a21d51735cf8667bcd610f057e88548d5d6a381401f6b4501a8bc6c1a9dc8498", "name_or_path": "nihalbaig/wav2vec2-large-xlsr-bn", "tokenizer_class": "Wav2Vec2CTCTokenizer"}
|
vocab.json
ADDED
@@ -0,0 +1 @@
|
|
|
|
|
1 |
+
{"=": 1, "।": 2, "ঁ": 3, "ং": 4, "ঃ": 5, "অ": 6, "আ": 7, "ই": 8, "ঈ": 9, "উ": 10, "ঊ": 11, "ঋ": 12, "এ": 13, "ঐ": 14, "ও": 15, "ঔ": 16, "ক": 17, "খ": 18, "গ": 19, "ঘ": 20, "ঙ": 21, "চ": 22, "ছ": 23, "জ": 24, "ঝ": 25, "ঞ": 26, "ট": 27, "ঠ": 28, "ড": 29, "ঢ": 30, "ণ": 31, "ত": 32, "থ": 33, "দ": 34, "ধ": 35, "ন": 36, "প": 37, "ফ": 38, "ব": 39, "ভ": 40, "ম": 41, "য": 42, "র": 43, "ল": 44, "শ": 45, "ষ": 46, "স": 47, "হ": 48, "া": 49, "ি": 50, "ী": 51, "ু": 52, "ূ": 53, "ৃ": 54, "ে": 55, "ৈ": 56, "ো": 57, "ৌ": 58, "্": 59, "ৎ": 60, "ড়": 61, "ঢ়": 62, "য়": 63, "০": 64, "১": 65, "২": 66, "৩": 67, "৪": 68, "৫": 69, "৬": 70, "৭": 71, "৮": 72, "৯": 73, "": 74, "—": 75, "|": 0, "[UNK]": 76, "[PAD]": 77}
|