LegolasTheElf
commited on
Commit
•
cb49f47
1
Parent(s):
281114e
Bengali Wav2Vec2 with LM !!!
Browse files- .gitattributes +1 -0
- alphabet.json +1 -0
- language_model/5gram.bin +3 -0
- language_model/attrs.json +1 -0
- language_model/unigrams.txt +3 -0
- tokenizer_config.json +1 -1
.gitattributes
CHANGED
@@ -25,3 +25,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
25 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
26 |
*.zstandard filter=lfs diff=lfs merge=lfs -text
|
27 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
25 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
26 |
*.zstandard filter=lfs diff=lfs merge=lfs -text
|
27 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
28 |
+
language_model/unigrams.txt filter=lfs diff=lfs merge=lfs -text
|
alphabet.json
ADDED
@@ -0,0 +1 @@
|
|
|
1 |
+
{"labels": ["\u09c8", "\u09b8", "\u09ad", "\u09eb", "\u09e7", "\u0981", "\u0986", "\u09f0", "\u0985", "\u0993", "\u09ea", "\u09a6", "\u0995", "\u09a1", "\u09ab", "\u09a0", "\u09dc", "\u09ec", "\u09ac", "\u0996", "\u09ed", "\u09ee", "\u099d", "\u09af", "\u0983", "\u09bc", "\u09be", "\u09e8", "\u09c7", "\u09b7", "\u09cd", "\u09b6", "\u09ae", "\u09df", "\u09a7", "\u098a", "\u09c3", " ", "\u09b2", "\u0987", "\u09a2", "\u09aa", "\u09ef", "\u0988", "\u098b", "\u0997", "\u09bf", "\u09c1", "\u09cb", "\u09b9", "\u09d7", "\u0998", "\u0982", "\u099e", "\u09e9", "\u09a4", "\u0999", "\u099c", "\u0990", "\u09c2", "\u09dd", "\u09b0", "\u0994", "\u09a5", "\u09ce", "\u0964", "\u099a", "\u09c0", "\u09e6", "\u09cc", "\u098f", "\u09a8", "\u09a3", "\u099f", "\u099b", "\u0989", "\u2047", "", "<s>", "</s>"], "is_bpe": false}
|
language_model/5gram.bin
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:93a6d4a176fff70071bf2d89ea723c5e4955dc8c2b2b98fbd8ee5196cd22fd2b
|
3 |
+
size 1658384767
|
language_model/attrs.json
ADDED
@@ -0,0 +1 @@
|
|
|
1 |
+
{"alpha": 0.5, "beta": 1.0, "unk_score_offset": -10.0, "score_boundary": true}
|
language_model/unigrams.txt
ADDED
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
1 |
+
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:aefc41cfe6c7ea3e80caba1cc72754c7bfca3f9a02581804298ac00e28c521fe
|
3 |
+
size 27964585
|
tokenizer_config.json
CHANGED
@@ -1 +1 @@
|
|
1 |
-
{"unk_token": "[UNK]", "bos_token": "<s>", "eos_token": "</s>", "pad_token": "[PAD]", "do_lower_case": false, "word_delimiter_token": " ", "
|
1 |
+
{"unk_token": "[UNK]", "bos_token": "<s>", "eos_token": "</s>", "pad_token": "[PAD]", "do_lower_case": false, "word_delimiter_token": " ", "processor_class": "Wav2Vec2Processor", "special_tokens_map_file": "/root/.cache/huggingface/transformers/615fa08d72a6825df33b3f770434f4add3cb02f0795cff28ab24417e789c07ed.a21d51735cf8667bcd610f057e88548d5d6a381401f6b4501a8bc6c1a9dc8498", "tokenizer_file": null, "name_or_path": "LegolasTheElf/Wav2Vec2_XLSR_Bengali_V2", "tokenizer_class": "Wav2Vec2CTCTokenizer"}
|