Add model files

Browse files

Files changed (9) hide show

.ipynb_checkpoints/README-checkpoint.md +144 -0
.ipynb_checkpoints/preprocessor_config-checkpoint.json +8 -0
.ipynb_checkpoints/special_tokens_map-checkpoint.json +1 -0
.ipynb_checkpoints/tokenizer_config-checkpoint.json +1 -0
.ipynb_checkpoints/vocab-checkpoint.json +1 -0
README.md +2 -2
config.json +1 -1
pytorch_model.bin +1 -1
vocab.json +1 -1

.ipynb_checkpoints/README-checkpoint.md ADDED Viewed

	@@ -0,0 +1,144 @@

+---
+language: ar
+datasets:
+- common_voice: Common Voice Corpus 4
+metrics:
+- wer
+tags:
+- audio
+- automatic-speech-recognition
+- speech
+- xlsr-fine-tuning-week
+license: apache-2.0
+model-index:
+- name: Hasni XLSR Wav2Vec2 Large 53
+  results:
+  - task:
+      name: Speech Recognition
+      type: automatic-speech-recognition
+    dataset:
+      name: Common Voice ar
+      type: common_voice
+      args: ar
+    metrics:
+       - name: Test WER
+         type: wer
+         value: 52.18
+---
+# Wav2Vec2-Large-XLSR-53-Arabic
+Fine-tuned [facebook/wav2vec2-large-xlsr-53](https://huggingface.co/facebook/wav2vec2-large-xlsr-53) on Arabic using the [Common Voice Corpus 4](https://commonvoice.mozilla.org/en/datasets) dataset.
+When using this model, make sure that your speech input is sampled at 16kHz.
+## Usage
+The model can be used directly (without a language model) as follows:
+```python
+import torch
+import torchaudio
+from datasets import load_dataset
+from transformers import Wav2Vec2ForCTC, Wav2Vec2Processor
+test_dataset = load_dataset("common_voice", "ar", split="test[:2%]")
+processor = Wav2Vec2Processor.from_pretrained("anas/wav2vec2-large-xlsr-arabic")
+model = Wav2Vec2ForCTC.from_pretrained("anas/wav2vec2-large-xlsr-arabic")
+resampler = torchaudio.transforms.Resample(48_000, 16_000)
+# Preprocessing the datasets.
+# We need to read the aduio files as arrays
+def speech_file_to_array_fn(batch):
+    speech_array, sampling_rate = torchaudio.load(batch["path"])
+    batch["speech"] = resampler(speech_array).squeeze().numpy()
+    return batch
+test_dataset = test_dataset.map(speech_file_to_array_fn)
+inputs = processor(test_dataset["speech"][:2], sampling_rate=16_000, return_tensors="pt", padding=True)
+with torch.no_grad():
+     logits = model(inputs.input_values, attention_mask=inputs.attention_mask).logits
+predicted_ids = torch.argmax(logits, dim=-1)
+print("Prediction:", processor.batch_decode(predicted_ids))
+print("Reference:", test_dataset["sentence"][:2])
+```
+## Evaluation
+The model can be evaluated as follows on the Arabic test data of Common Voice.
+```python
+import torch
+import torchaudio
+from datasets import load_dataset, load_metric
+from transformers import Wav2Vec2ForCTC, Wav2Vec2Processor
+import re
+test_dataset = load_dataset("common_voice", "ar", split="test")
+processor = Wav2Vec2Processor.from_pretrained("anas/wav2vec2-large-xlsr-arabic")
+model = Wav2Vec2ForCTC.from_pretrained("anas/wav2vec2-large-xlsr-arabic/")
+model.to("cuda")
+chars_to_ignore_regex = '[\\\\,\\\\؟\\\\.\\\\!\\\\-\\\\;\\\\\\\\:\\\\'\\\\"\\\\☭\\\\«\\\\»\\\\؛\\\\—\\\\ـ\\\\_\\\\،\\\\“\\\\%\\\\‘\\\\”\\\\�]'
+resampler = torchaudio.transforms.Resample(48_000, 16_000)
+# Preprocessing the datasets.
+# We need to read the aduio files as arrays
+def speech_file_to_array_fn(batch):
+    batch["sentence"] = re.sub(chars_to_ignore_regex, '', batch["sentence"]).lower()
+    batch["sentence"] = re.sub('[a-z]','',batch["sentence"])
+    batch["sentence"] = re.sub("[إأٱآا]", "ا", batch["sentence"])
+    noise = re.compile(""" ّ    | # Tashdid
+                             َ    | # Fatha
+                             ً    | # Tanwin Fath
+                             ُ    | # Damma
+                             ٌ    | # Tanwin Damm
+                             ِ    | # Kasra
+                             ٍ    | # Tanwin Kasr
+                             ْ    | # Sukun
+                             ـ     # Tatwil/Kashida
+                         """, re.VERBOSE)
+    batch["sentence"] = re.sub(noise, '', batch["sentence"])
+    speech_array, sampling_rate = torchaudio.load(batch["path"])
+    batch["speech"] = resampler(speech_array).squeeze().numpy()
+    return batch
+test_dataset = test_dataset.map(speech_file_to_array_fn)
+# Preprocessing the datasets.
+# We need to read the aduio files as arrays
+def evaluate(batch):
+    inputs = processor(batch["speech"], sampling_rate=16_000, return_tensors="pt", padding=True)
+    with torch.no_grad():
+         logits = model(inputs.input_values.to("cuda"), attention_mask=inputs.attention_mask.to("cuda")).logits
+         pred_ids = torch.argmax(logits, dim=-1)
+         batch["pred_strings"] = processor.batch_decode(pred_ids)
+    return batch
+result = test_dataset.map(evaluate, batched=True, batch_size=8)
+print("WER: {:2f}".format(100 * wer.compute(predictions=result["pred_strings"], references=result["sentence"])))
+```
+**Test Result**: 52.18 %
+## Training
+The Common Voice Corpus 4 `train`, `validation`, datasets were used for training
+The script used for training can be found [here](...)
+Twitter: [here](https://twitter.com/hasnii_anas)
+Email: anashasni146@gmail.com

.ipynb_checkpoints/preprocessor_config-checkpoint.json ADDED Viewed

	@@ -0,0 +1,8 @@

+{
+  "do_normalize": true,
+  "feature_size": 1,
+  "padding_side": "right",
+  "padding_value": 0.0,
+  "return_attention_mask": true,
+  "sampling_rate": 16000
+}

.ipynb_checkpoints/special_tokens_map-checkpoint.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"bos_token": "<s>", "eos_token": "</s>", "unk_token": "[UNK]", "pad_token": "[PAD]"}

.ipynb_checkpoints/tokenizer_config-checkpoint.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"unk_token": "[UNK]", "bos_token": "<s>", "eos_token": "</s>", "pad_token": "[PAD]", "do_lower_case": false, "word_delimiter_token": "\|"}

.ipynb_checkpoints/vocab-checkpoint.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"خ": 0, "ة": 1, "د": 2, "ا": 4, "ض": 5, "م": 6, "و": 7, "ك": 8, "ث": 9, "ش": 10, "ع": 11, "ز": 12, "ء": 13, "ی": 14, "ن": 15, "ه": 16, "ق": 17, "ت": 18, "ب": 19, "ف": 20, "ظ": 21, "ح": 22, "ص": 23, "ئ": 24, "ذ": 25, "ى": 26, "غ": 27, "س": 28, "ر": 29, "ط": 30, "ي": 31, "ل": 32, "ؤ": 33, "ج": 34, "\|": 3, "[UNK]": 35, "[PAD]": 36}

README.md CHANGED Viewed

@@ -23,7 +23,7 @@ model-index:
     metrics:
        - name: Test WER
          type: wer
-         value: 59.67
 ---
 # Wav2Vec2-Large-XLSR-53-Arabic
@@ -130,7 +130,7 @@ result = test_dataset.map(evaluate, batched=True, batch_size=8)
 print("WER: {:2f}".format(100 * wer.compute(predictions=result["pred_strings"], references=result["sentence"])))
 ```
-**Test Result**: 59.67 %
 ## Training

     metrics:
        - name: Test WER
          type: wer
+         value: 52.18
 ---
 # Wav2Vec2-Large-XLSR-53-Arabic
 print("WER: {:2f}".format(100 * wer.compute(predictions=result["pred_strings"], references=result["sentence"])))
 ```
+**Test Result**: 52.18 %
 ## Training

config.json CHANGED Viewed

@@ -46,7 +46,7 @@
   "final_dropout": 0.0,
   "gradient_checkpointing": true,
   "hidden_act": "gelu",
-  "hidden_dropout": 0.1,
   "hidden_size": 1024,
   "initializer_range": 0.02,
   "intermediate_size": 4096,

   "final_dropout": 0.0,
   "gradient_checkpointing": true,
   "hidden_act": "gelu",
+  "hidden_dropout": 0.05,
   "hidden_size": 1024,
   "initializer_range": 0.02,
   "intermediate_size": 4096,

pytorch_model.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:70dc8c441b2f93e47a9a02b7dd3ceae28dd595875c017c98a18f0d7d4e7d7f43
 size 1262085527

 version https://git-lfs.github.com/spec/v1
+oid sha256:c9d92c7e4e59488cb3de5cb0336893c24517ea0da99be9f9cd6f77ada2ecbe0b
 size 1262085527

vocab.json CHANGED Viewed

	@@ -1 +1 @@
1	- {"ن": 0, "م": 1, "ش": 2, "د": ~~3, "ف":~~ 4, "خ": 5, "س": 6, "ك": 7, "ض": 8, "ؤ": 9, "ط": 10, "ء": 11, "ص": 12, "ی": 13, "ل": 14, "ظ": 15, "ه": 16, "ب": 17, "غ": 18, "ح": 19, "ث": 20, "ة": 21, "ي": 22, "ت": 23, "ى": 24, "ج": 25, "ق": 26, "ر": 27, "ا": 28, "ع": 29, "ذ": 30, "ز": 31, "ئ": 32, "و": 34, "\|": 33, "[UNK]": 35, "[PAD]": 36}


1	+ {"خ": 0, "ة": 1, "د": 2, "ا": 4, "ض": 5, "م": 6, "و": 7, "ك": 8, "ث": 9, "ش": 10, "ع": 11, "ز": 12, "ء": 13, "ی": 14, "ن": 15, "ه": 16, "ق": 17, "ت": 18, "ب": 19, "ف": 20, "ظ": 21, "ح": 22, "ص": 23, "ئ": 24, "ذ": 25, "ى": 26, "غ": 27, "س": 28, "ر": 29, "ط": 30, "ي": 31, "ل": 32, "ؤ": 33, "ج": 34, "\|": 3, "[UNK]": 35, "[PAD]": 36}