Spaces:

k2-fsa
/

automatic-speech-recognition

Running

App Files Files Community

csukuangfj commited on Jul 17, 2022

Commit

6b31279

•

1 Parent(s): 500c811

small fixes

Browse files

Files changed (2) hide show

app.py +63 -9
model.py +49 -0

app.py CHANGED Viewed

@@ -16,14 +16,68 @@
 # See the License for the specific language governing permissions and
 # limitations under the License.
 import gradio as gr
 demo = gr.Blocks()
-def process_uploaded_file(uploaded_file: str):
-    print("uploaded_file", uploaded_file)
-    return "hello"
 with demo:
@@ -36,9 +90,9 @@ with demo:
                 optional=False,
                 label="Upload from disk",
             )
-            upload_button = gr.Button("Upload")
             uploaded_output = gr.outputs.Textbox(
-                label="Recognized speech for uploaded file"
             )
         with gr.TabItem("Record from microphone"):
@@ -49,18 +103,18 @@ with demo:
                 label="Record from microphone",
             )
             recorded_output = gr.outputs.Textbox(
-                label="Recognized speech for recordings"
             )
-            record_button = gr.Button("Record")
         upload_button.click(
-            process_uploaded_file,
             inputs=uploaded_file,
             outputs=uploaded_output,
         )
         record_button.click(
-            process_uploaded_file,
             inputs=microphone,
             outputs=recorded_output,
         )

 # See the License for the specific language governing permissions and
 # limitations under the License.
+import os
+import time
+from datetime import datetime
 import gradio as gr
+import torchaudio
+from model import get_gigaspeech_pre_trained_model, sample_rate
+models = {"english": get_gigaspeech_pre_trained_model()}
+def convert_to_wav(in_filename: str) -> str:
+    """Convert the input audio file to a wave file"""
+    out_filename = in_filename + ".wav"
+    print(f"Converting '{in_filename}' to '{out_filename}'")
+    _ = os.system(f"ffmpeg -hide_banner -i '{in_filename}' '{out_filename}'")
+    return out_filename
 demo = gr.Blocks()
+def process(in_filename: str) -> str:
+    print("in_filename", in_filename)
+    filename = convert_to_wav(in_filename)
+    now = datetime.now()
+    date_time = now.strftime("%Y-%m-%d %H:%M:%S.%f")
+    print(f"Started at {date_time}")
+    start = time.time()
+    wave, wave_sample_rate = torchaudio.load(filename)
+    if wave_sample_rate != sample_rate:
+        print(
+            f"Expected sample rate: {sample_rate}. Given: {wave_sample_rate}. "
+            f"Resampling to {sample_rate}."
+        )
+        wave = torchaudio.functional.resample(
+            wave,
+            orig_freq=wave_sample_rate,
+            new_freq=sample_rate,
+        )
+    wave = wave[0]  # use only the first channel.
+    hyp = models["english"].decode_waves([wave])[0]
+    date_time = now.strftime("%Y-%m-%d %H:%M:%S.%f")
+    end = time.time()
+    duration = wave.shape[0] / sample_rate
+    rtf = (end - start) / duration
+    print(f"Finished at {date_time} s. Elapsed: {end - start: .3f} s")
+    print(f"Duration {duration: .3f} s")
+    print(f"RTF {rtf: .3f}")
+    print("hyp")
+    print(hyp)
+    return hyp
 with demo:
                 optional=False,
                 label="Upload from disk",
             )
+            upload_button = gr.Button("Submit for recognition")
             uploaded_output = gr.outputs.Textbox(
+                label="Recognized speech from uploaded file"
             )
         with gr.TabItem("Record from microphone"):
                 label="Record from microphone",
             )
             recorded_output = gr.outputs.Textbox(
+                label="Recognized speech from recordings"
             )
+            record_button = gr.Button("Submit for recordings")
         upload_button.click(
+            process,
             inputs=uploaded_file,
             outputs=uploaded_output,
         )
         record_button.click(
+            process,
             inputs=microphone,
             outputs=recorded_output,
         )

model.py ADDED Viewed

	@@ -0,0 +1,49 @@

+# Copyright      2022  Xiaomi Corp.        (authors: Fangjun Kuang)
+#
+# See LICENSE for clarification regarding multiple authors
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+from huggingface_hub import hf_hub_download
+from functools import lru_cache
+from offline_asr import OfflineAsr
+sample_rate = 16000
+@lru_cache(maxsize=1)
+def get_gigaspeech_pre_trained_model():
+    nn_model_filename = hf_hub_download(
+        # It is converted from https://huggingface.co/wgb14/icefall-asr-gigaspeech-pruned-transducer-stateless2
+        repo_id="csukuangfj/icefall-asr-gigaspeech-pruned-transducer-stateless2",
+        filename="cpu_jit-epoch-29-avg-11-torch-1.10.0.pt",
+        subfolder="exp",
+    )
+    bpe_model_filename = hf_hub_download(
+        repo_id="wgb14/icefall-asr-gigaspeech-pruned-transducer-stateless2",
+        filename="bpe.model",
+        subfolder="data/lang_bpe_500",
+    )
+    return OfflineAsr(
+        nn_model_filename=nn_model_filename,
+        bpe_model_filename=bpe_model_filename,
+        token_filename=None,
+        decoding_method="greedy_search",
+        num_active_paths=4,
+        sample_rate=sample_rate,
+        device="cpu",
+    )