LegolasTheElf commited on
Commit
cb49f47
1 Parent(s): 281114e

Bengali Wav2Vec2 with LM !!!

Browse files
.gitattributes CHANGED
@@ -25,3 +25,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
25
  *.zip filter=lfs diff=lfs merge=lfs -text
26
  *.zstandard filter=lfs diff=lfs merge=lfs -text
27
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
25
  *.zip filter=lfs diff=lfs merge=lfs -text
26
  *.zstandard filter=lfs diff=lfs merge=lfs -text
27
  *tfevents* filter=lfs diff=lfs merge=lfs -text
28
+ language_model/unigrams.txt filter=lfs diff=lfs merge=lfs -text
alphabet.json ADDED
@@ -0,0 +1 @@
 
1
+ {"labels": ["\u09c8", "\u09b8", "\u09ad", "\u09eb", "\u09e7", "\u0981", "\u0986", "\u09f0", "\u0985", "\u0993", "\u09ea", "\u09a6", "\u0995", "\u09a1", "\u09ab", "\u09a0", "\u09dc", "\u09ec", "\u09ac", "\u0996", "\u09ed", "\u09ee", "\u099d", "\u09af", "\u0983", "\u09bc", "\u09be", "\u09e8", "\u09c7", "\u09b7", "\u09cd", "\u09b6", "\u09ae", "\u09df", "\u09a7", "\u098a", "\u09c3", " ", "\u09b2", "\u0987", "\u09a2", "\u09aa", "\u09ef", "\u0988", "\u098b", "\u0997", "\u09bf", "\u09c1", "\u09cb", "\u09b9", "\u09d7", "\u0998", "\u0982", "\u099e", "\u09e9", "\u09a4", "\u0999", "\u099c", "\u0990", "\u09c2", "\u09dd", "\u09b0", "\u0994", "\u09a5", "\u09ce", "\u0964", "\u099a", "\u09c0", "\u09e6", "\u09cc", "\u098f", "\u09a8", "\u09a3", "\u099f", "\u099b", "\u0989", "\u2047", "", "<s>", "</s>"], "is_bpe": false}
language_model/5gram.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:93a6d4a176fff70071bf2d89ea723c5e4955dc8c2b2b98fbd8ee5196cd22fd2b
3
+ size 1658384767
language_model/attrs.json ADDED
@@ -0,0 +1 @@
 
1
+ {"alpha": 0.5, "beta": 1.0, "unk_score_offset": -10.0, "score_boundary": true}
language_model/unigrams.txt ADDED
@@ -0,0 +1,3 @@
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:aefc41cfe6c7ea3e80caba1cc72754c7bfca3f9a02581804298ac00e28c521fe
3
+ size 27964585
tokenizer_config.json CHANGED
@@ -1 +1 @@
1
- {"unk_token": "[UNK]", "bos_token": "<s>", "eos_token": "</s>", "pad_token": "[PAD]", "do_lower_case": false, "word_delimiter_token": " ", "tokenizer_class": "Wav2Vec2CTCTokenizer", "processor_class": "Wav2Vec2Processor"}
1
+ {"unk_token": "[UNK]", "bos_token": "<s>", "eos_token": "</s>", "pad_token": "[PAD]", "do_lower_case": false, "word_delimiter_token": " ", "processor_class": "Wav2Vec2Processor", "special_tokens_map_file": "/root/.cache/huggingface/transformers/615fa08d72a6825df33b3f770434f4add3cb02f0795cff28ab24417e789c07ed.a21d51735cf8667bcd610f057e88548d5d6a381401f6b4501a8bc6c1a9dc8498", "tokenizer_file": null, "name_or_path": "LegolasTheElf/Wav2Vec2_XLSR_Bengali_V2", "tokenizer_class": "Wav2Vec2CTCTokenizer"}