YingxuHe commited on
Commit
8213805
·
verified ·
1 Parent(s): a083b75

Upload folder using huggingface_hub

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +80 -0
  2. .gitignore +0 -13
  3. .pre-commit-config.yaml +0 -53
  4. Makefile +0 -13
  5. README.md +2 -38
  6. __init__.py +0 -0
  7. additional_info/Leaderboard-Rename.xlsx +0 -0
  8. additional_info/__init__.py +0 -0
  9. app.py +0 -112
  10. app/__init__.py +0 -0
  11. app/content.py +0 -431
  12. app/draw_diagram.py +0 -193
  13. app/pages.py +0 -576
  14. app/show_examples.py +0 -199
  15. app/summarization.py +0 -133
  16. assets/index-C9x3RgO7.js +0 -0
  17. assets/index-D38mSSRC.css +1 -0
  18. examples/cna_test/example_0.wav +0 -0
  19. examples/{ytb_asr_batch3_tamil/data-00000-of-00001.arrow → cna_test/example_1.wav} +2 -2
  20. examples/{ytb_sds_batch3_chinese/data-00000-of-00001.arrow → commonvoice_17_id_asr/example_0.wav} +2 -2
  21. examples/{ytb_asr_batch3_chinese/data-00000-of-00001.arrow → commonvoice_17_id_asr/example_1.wav} +2 -2
  22. examples/{ytb_asr_batch3_malay/data-00000-of-00001.arrow → commonvoice_17_ta_asr/example_0.wav} +2 -2
  23. examples/commonvoice_17_ta_asr/example_1.wav +3 -0
  24. examples/commonvoice_17_th_asr/example_0.wav +0 -0
  25. examples/commonvoice_17_th_asr/example_1.wav +3 -0
  26. examples/commonvoice_17_vi_asr/example_0.wav +3 -0
  27. examples/commonvoice_17_vi_asr/example_1.wav +3 -0
  28. examples/commonvoice_zh_asr/example_0.wav +3 -0
  29. examples/commonvoice_zh_asr/example_1.wav +3 -0
  30. examples/fleurs_tamil_ta_30_asr/example_0.wav +3 -0
  31. examples/fleurs_tamil_ta_30_asr/example_1.wav +3 -0
  32. examples/gigaspeech2_id_test/example_0.wav +3 -0
  33. examples/gigaspeech2_id_test/example_1.wav +0 -0
  34. examples/gigaspeech2_th_test/example_0.wav +3 -0
  35. examples/gigaspeech2_th_test/example_1.wav +3 -0
  36. examples/gigaspeech2_vi_test/example_0.wav +3 -0
  37. examples/gigaspeech2_vi_test/example_1.wav +3 -0
  38. examples/idpc_short_test/example_0.wav +3 -0
  39. examples/idpc_short_test/example_1.wav +3 -0
  40. examples/imda_ar_dialogue/example_0.wav +3 -0
  41. examples/imda_ar_dialogue/example_1.wav +3 -0
  42. examples/imda_ar_sentence/example_0.wav +3 -0
  43. examples/imda_ar_sentence/example_1.wav +3 -0
  44. examples/imda_part1_asr_test/example_0.wav +3 -0
  45. examples/imda_part1_asr_test/example_1.wav +3 -0
  46. examples/imda_part2_asr_test/example_0.wav +3 -0
  47. examples/imda_part2_asr_test/example_1.wav +3 -0
  48. examples/imda_part3_30s_asr_test/example_0.wav +3 -0
  49. examples/imda_part3_30s_asr_test/example_1.wav +3 -0
  50. examples/imda_part3_30s_ds_human_test/example_0.wav +3 -0
.gitattributes CHANGED
@@ -79,3 +79,83 @@ examples/2SQA/Spoken-Squad-Test/sample_1.wav filter=lfs diff=lfs merge=lfs -text
79
  examples/2SQA/Spoken-Squad-Test/sample_2.wav filter=lfs diff=lfs merge=lfs -text
80
  examples/SQA/DREAM-TTS-MCQ-Test/sample_1.wav filter=lfs diff=lfs merge=lfs -text
81
  examples/**/*.arrow filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
79
  examples/2SQA/Spoken-Squad-Test/sample_2.wav filter=lfs diff=lfs merge=lfs -text
80
  examples/SQA/DREAM-TTS-MCQ-Test/sample_1.wav filter=lfs diff=lfs merge=lfs -text
81
  examples/**/*.arrow filter=lfs diff=lfs merge=lfs -text
82
+ examples/cna_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
83
+ examples/commonvoice_17_id_asr/example_0.wav filter=lfs diff=lfs merge=lfs -text
84
+ examples/commonvoice_17_id_asr/example_1.wav filter=lfs diff=lfs merge=lfs -text
85
+ examples/commonvoice_17_ta_asr/example_0.wav filter=lfs diff=lfs merge=lfs -text
86
+ examples/commonvoice_17_ta_asr/example_1.wav filter=lfs diff=lfs merge=lfs -text
87
+ examples/commonvoice_17_th_asr/example_1.wav filter=lfs diff=lfs merge=lfs -text
88
+ examples/commonvoice_17_vi_asr/example_0.wav filter=lfs diff=lfs merge=lfs -text
89
+ examples/commonvoice_17_vi_asr/example_1.wav filter=lfs diff=lfs merge=lfs -text
90
+ examples/commonvoice_zh_asr/example_0.wav filter=lfs diff=lfs merge=lfs -text
91
+ examples/commonvoice_zh_asr/example_1.wav filter=lfs diff=lfs merge=lfs -text
92
+ examples/fleurs_tamil_ta_30_asr/example_0.wav filter=lfs diff=lfs merge=lfs -text
93
+ examples/fleurs_tamil_ta_30_asr/example_1.wav filter=lfs diff=lfs merge=lfs -text
94
+ examples/gigaspeech2_id_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
95
+ examples/gigaspeech2_th_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
96
+ examples/gigaspeech2_th_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
97
+ examples/gigaspeech2_vi_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
98
+ examples/gigaspeech2_vi_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
99
+ examples/idpc_short_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
100
+ examples/idpc_short_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
101
+ examples/imda_ar_dialogue/example_0.wav filter=lfs diff=lfs merge=lfs -text
102
+ examples/imda_ar_dialogue/example_1.wav filter=lfs diff=lfs merge=lfs -text
103
+ examples/imda_ar_sentence/example_0.wav filter=lfs diff=lfs merge=lfs -text
104
+ examples/imda_ar_sentence/example_1.wav filter=lfs diff=lfs merge=lfs -text
105
+ examples/imda_part1_asr_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
106
+ examples/imda_part1_asr_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
107
+ examples/imda_part2_asr_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
108
+ examples/imda_part2_asr_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
109
+ examples/imda_part3_30s_asr_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
110
+ examples/imda_part3_30s_asr_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
111
+ examples/imda_part3_30s_ds_human_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
112
+ examples/imda_part3_30s_ds_human_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
113
+ examples/imda_part3_30s_sqa_human_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
114
+ examples/imda_part3_30s_sqa_human_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
115
+ examples/imda_part4_30s_asr_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
116
+ examples/imda_part4_30s_asr_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
117
+ examples/imda_part4_30s_ds_human_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
118
+ examples/imda_part4_30s_ds_human_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
119
+ examples/imda_part4_30s_sqa_human_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
120
+ examples/imda_part4_30s_sqa_human_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
121
+ examples/imda_part5_30s_asr_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
122
+ examples/imda_part5_30s_asr_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
123
+ examples/imda_part5_30s_ds_human_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
124
+ examples/imda_part5_30s_ds_human_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
125
+ examples/imda_part5_30s_sqa_human_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
126
+ examples/imda_part5_30s_sqa_human_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
127
+ examples/imda_part6_30s_asr_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
128
+ examples/imda_part6_30s_asr_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
129
+ examples/imda_part6_30s_ds_human_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
130
+ examples/imda_part6_30s_ds_human_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
131
+ examples/imda_part6_30s_sqa_human_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
132
+ examples/imda_part6_30s_sqa_human_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
133
+ examples/lotus_thai_th_30_asr/example_0.wav filter=lfs diff=lfs merge=lfs -text
134
+ examples/lotus_thai_th_30_asr/example_1.wav filter=lfs diff=lfs merge=lfs -text
135
+ examples/mediacorp_short_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
136
+ examples/mediacorp_short_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
137
+ examples/mediacorp_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
138
+ examples/mediacorp_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
139
+ examples/parliament_short_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
140
+ examples/parliament_short_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
141
+ examples/seame_dev_sge/example_0.wav filter=lfs diff=lfs merge=lfs -text
142
+ examples/ukusnews_short_test/example_0.wav filter=lfs diff=lfs merge=lfs -text
143
+ examples/ukusnews_short_test/example_1.wav filter=lfs diff=lfs merge=lfs -text
144
+ examples/ytb_asr_batch1/example_0.wav filter=lfs diff=lfs merge=lfs -text
145
+ examples/ytb_asr_batch1/example_1.wav filter=lfs diff=lfs merge=lfs -text
146
+ examples/ytb_asr_batch2/example_0.wav filter=lfs diff=lfs merge=lfs -text
147
+ examples/ytb_asr_batch2/example_1.wav filter=lfs diff=lfs merge=lfs -text
148
+ examples/ytb_asr_batch3_chinese/example_0.wav filter=lfs diff=lfs merge=lfs -text
149
+ examples/ytb_asr_batch3_chinese/example_1.wav filter=lfs diff=lfs merge=lfs -text
150
+ examples/ytb_asr_batch3_malay/example_0.wav filter=lfs diff=lfs merge=lfs -text
151
+ examples/ytb_asr_batch3_malay/example_1.wav filter=lfs diff=lfs merge=lfs -text
152
+ examples/ytb_asr_batch3_tamil/example_0.wav filter=lfs diff=lfs merge=lfs -text
153
+ examples/ytb_asr_batch3_tamil/example_1.wav filter=lfs diff=lfs merge=lfs -text
154
+ examples/ytb_asr_batch3_tamil_v2/example_0.wav filter=lfs diff=lfs merge=lfs -text
155
+ examples/ytb_asr_batch3_tamil_v2/example_1.wav filter=lfs diff=lfs merge=lfs -text
156
+ examples/ytb_pqa_batch1/example_0.wav filter=lfs diff=lfs merge=lfs -text
157
+ examples/ytb_pqa_batch1/example_1.wav filter=lfs diff=lfs merge=lfs -text
158
+ examples/ytb_sds_batch1/example_0.wav filter=lfs diff=lfs merge=lfs -text
159
+ examples/ytb_sds_batch1/example_1.wav filter=lfs diff=lfs merge=lfs -text
160
+ examples/ytb_sqa_batch1/example_0.wav filter=lfs diff=lfs merge=lfs -text
161
+ examples/ytb_sqa_batch1/example_1.wav filter=lfs diff=lfs merge=lfs -text
.gitignore DELETED
@@ -1,13 +0,0 @@
1
- auto_evals/
2
- venv/
3
- __pycache__/
4
- .env
5
- .ipynb_checkpoints
6
- *ipynb
7
- .vscode/
8
-
9
- eval-queue/
10
- eval-results/
11
- eval-queue-bk/
12
- eval-results-bk/
13
- logs/
 
 
 
 
 
 
 
 
 
 
 
 
 
 
.pre-commit-config.yaml DELETED
@@ -1,53 +0,0 @@
1
- # Copyright (c) 2022, NVIDIA CORPORATION. All rights reserved.
2
- #
3
- # Licensed under the Apache License, Version 2.0 (the "License");
4
- # you may not use this file except in compliance with the License.
5
- # You may obtain a copy of the License at
6
- #
7
- # http://www.apache.org/licenses/LICENSE-2.0
8
- #
9
- # Unless required by applicable law or agreed to in writing, software
10
- # distributed under the License is distributed on an "AS IS" BASIS,
11
- # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
- # See the License for the specific language governing permissions and
13
- # limitations under the License.
14
-
15
- default_language_version:
16
- python: python3
17
-
18
- ci:
19
- autofix_prs: true
20
- autoupdate_commit_msg: '[pre-commit.ci] pre-commit suggestions'
21
- autoupdate_schedule: quarterly
22
-
23
- repos:
24
- - repo: https://github.com/pre-commit/pre-commit-hooks
25
- rev: v4.3.0
26
- hooks:
27
- - id: check-yaml
28
- - id: check-case-conflict
29
- - id: detect-private-key
30
- - id: check-added-large-files
31
- args: ['--maxkb=1000']
32
- - id: requirements-txt-fixer
33
- - id: end-of-file-fixer
34
- - id: trailing-whitespace
35
-
36
- - repo: https://github.com/PyCQA/isort
37
- rev: 5.12.0
38
- hooks:
39
- - id: isort
40
- name: Format imports
41
-
42
- - repo: https://github.com/psf/black
43
- rev: 22.12.0
44
- hooks:
45
- - id: black
46
- name: Format code
47
- additional_dependencies: ['click==8.0.2']
48
-
49
- - repo: https://github.com/charliermarsh/ruff-pre-commit
50
- # Ruff version.
51
- rev: 'v0.0.267'
52
- hooks:
53
- - id: ruff
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
Makefile DELETED
@@ -1,13 +0,0 @@
1
- .PHONY: style format
2
-
3
-
4
- style:
5
- python -m black --line-length 119 .
6
- python -m isort .
7
- ruff check --fix .
8
-
9
-
10
- quality:
11
- python -m black --check --line-length 119 .
12
- python -m isort --check-only .
13
- ruff check .
 
 
 
 
 
 
 
 
 
 
 
 
 
 
README.md CHANGED
@@ -1,45 +1,9 @@
1
  ---
2
- title: Leaderboard / AudioBench
3
  emoji: 🥇
4
  colorFrom: green
5
  colorTo: indigo
6
- sdk: streamlit
7
- app_file: app.py
8
  pinned: false
9
- sdk_version: 1.37.1
10
  license: apache-2.0
11
  ---
12
-
13
- # Start the configuration
14
-
15
- Most of the variables to change for a default leaderboard are in `src/env.py` (replace the path for your leaderboard) and `src/about.py` (for tasks).
16
-
17
- Results files should have the following format and be stored as json files:
18
- ```json
19
- {
20
- "config": {
21
- "model_dtype": "torch.float16", # or torch.bfloat16 or 8bit or 4bit
22
- "model_name": "path of the model on the hub: org/model",
23
- "model_sha": "revision on the hub",
24
- },
25
- "results": {
26
- "task_name": {
27
- "metric_name": score,
28
- },
29
- "task_name2": {
30
- "metric_name": score,
31
- }
32
- }
33
- }
34
- ```
35
-
36
- Request files are created automatically by this tool.
37
-
38
- If you encounter problem on the space, don't hesitate to restart it to remove the create eval-queue, eval-queue-bk, eval-results and eval-results-bk created folder.
39
-
40
- # Code logic for more complex edits
41
-
42
- You'll find
43
- - the main table' columns names and properties in `src/display/utils.py`
44
- - the logic to read all results and request files, then convert them in dataframe lines, in `src/leaderboard/read_evals.py`, and `src/populate.py`
45
- - teh logic to allow or filter submissions in `src/submission/submit.py` and `src/submission/check_validity.py`
 
1
  ---
2
+ title: AudioBench Leaderboard
3
  emoji: 🥇
4
  colorFrom: green
5
  colorTo: indigo
6
+ sdk: static
 
7
  pinned: false
 
8
  license: apache-2.0
9
  ---
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
__init__.py DELETED
File without changes
additional_info/Leaderboard-Rename.xlsx DELETED
Binary file (13 kB)
 
additional_info/__init__.py DELETED
File without changes
app.py DELETED
@@ -1,112 +0,0 @@
1
- import streamlit as st
2
- import streamlit_antd_components as sac
3
-
4
- from app.pages import *
5
-
6
-
7
- # Set page configuration
8
- st.set_page_config(
9
- page_title="AudioBench Leaderboard",
10
- page_icon=":chart_with_upwards_trend:",
11
- layout="wide",
12
- )
13
-
14
-
15
-
16
- # Dictionary mapping menu items to their corresponding functions
17
- pages = {
18
- 'Dashboard' : dashboard,
19
- 'ASR-English' : asr_english,
20
- 'ASR-Mandarin' : asr_mandarin,
21
- 'ASR-Singlish' : asr_singlish,
22
- 'ASR-Malay' : asr_malay,
23
- 'ASR-Tamil' : asr_tamil,
24
- 'ASR-Indonesian' : asr_indonesian,
25
- 'ASR-Thai' : asr_thai,
26
- 'ASR-Vietnamese' : asr_vietnamese,
27
- 'ASR-Other' : asr_private,
28
- 'Speech Translation' : speech_translation,
29
- 'SQA-English' : speech_question_answering_english,
30
- 'SQA-Singlish' : speech_question_answering_singlish,
31
- 'SDS-Singlish' : spoken_dialogue_summarization_singlish,
32
- 'Speech Instruction' : speech_instruction,
33
- 'Audio Captioning' : audio_captioning,
34
- 'Audio-Scene QA' : audio_scene_question_answering,
35
- 'Accent Recognition' : accent_recognition,
36
- 'Gender Recognition' : gender_recognition,
37
- 'Emotion Recognition': emotion_recognition,
38
- 'Music Understanding': music_understanding,
39
-
40
- '* Under Development *': under_development,
41
- }
42
-
43
- # Initialize session state for menu selection
44
- if 'selected_menu' not in st.session_state:
45
- st.session_state.selected_menu = 'Introduction'
46
-
47
- # Define the menu items
48
- menu_items = [
49
- sac.MenuItem(label='Dashboard', icon='house'),
50
-
51
- sac.MenuItem(label='Automatic Speech Recognition', icon='mic',
52
- children = [
53
- sac.MenuItem(label='ASR-English', icon='mic'),
54
- sac.MenuItem(label='ASR-Singlish', icon='mic'),
55
- sac.MenuItem(label='ASR-Mandarin', icon='mic'),
56
- sac.MenuItem(label='ASR-Malay', icon='mic'),
57
- sac.MenuItem(label='ASR-Tamil', icon='mic'),
58
- sac.MenuItem(label='ASR-Indonesian', icon='mic'),
59
- sac.MenuItem(label='ASR-Thai', icon='mic'),
60
- sac.MenuItem(label='ASR-Vietnamese', icon='mic'),
61
- sac.MenuItem(label='ASR-Other', icon='mic'),
62
- ]
63
- ),
64
-
65
- sac.MenuItem(label='Speech Translation', icon='translate'
66
- ),
67
-
68
- sac.MenuItem(label='Spoken Question Answering', icon='question-circle',
69
- children = [
70
- sac.MenuItem(label='SQA-English', icon='mic'),
71
- sac.MenuItem(label='SQA-Singlish', icon='mic'),
72
- ]
73
- ),
74
-
75
- sac.MenuItem(label='Spoken Dialogue Summarization', icon='question-circle',
76
- children = [
77
- sac.MenuItem(label='SDS-Singlish', icon='mic'),
78
- ]
79
- ),
80
-
81
- sac.MenuItem(label='Speech Instruction', icon='mic-fill'),
82
-
83
- sac.MenuItem(label='Audio Captioning', icon='volume-down'),
84
-
85
- sac.MenuItem(label='Audio-Scene QA', icon='question-diamond-fill'),
86
-
87
- sac.MenuItem(label='Accent Recognition', icon='person-badge-fill'),
88
-
89
- sac.MenuItem(label='Gender Recognition', icon='gender-ambiguous'),
90
-
91
- sac.MenuItem(label='Emotion Recognition', icon='emoji-smile-fill'),
92
-
93
- sac.MenuItem(label='Music Understanding', icon='music-note-list'),
94
-
95
- sac.MenuItem(label='* Under Development *', icon='lock'),
96
-
97
- ]
98
-
99
- # Render the menu in the sidebar
100
- with st.sidebar:
101
- selected = sac.menu(menu_items,
102
- size='sm',
103
- open_all=False,
104
- )
105
-
106
- # Update session state based on selection
107
- if selected:
108
- st.session_state.selected_menu = selected
109
-
110
- # Display the selected page's content
111
- page = pages[st.session_state.selected_menu]
112
- page()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
app/__init__.py DELETED
File without changes
app/content.py DELETED
@@ -1,431 +0,0 @@
1
- asr_english_datasets = [
2
- 'LibriSpeech-Clean',
3
- 'LibriSpeech-Other',
4
- 'CommonVoice-15-EN',
5
- 'Peoples-Speech',
6
- 'GigaSpeech-1',
7
- 'Earnings-21',
8
- 'Earnings-22',
9
- 'TED-LIUM-3',
10
- 'TED-LIUM-3-LongForm',
11
- ]
12
-
13
-
14
- asr_singlish_datasets = [
15
- 'MNSC-PART1-ASR',
16
- 'MNSC-PART2-ASR',
17
- 'MNSC-PART3-ASR',
18
- 'MNSC-PART4-ASR',
19
- 'MNSC-PART5-ASR',
20
- 'MNSC-PART6-ASR',
21
- ]
22
-
23
-
24
- asr_mandarin_datasets = [
25
- 'AISHELL-ASR-ZH',
26
- 'CommonVoice-ZH',
27
- 'YouTube ASR: Chinese with English Prompt',
28
- ]
29
-
30
-
31
- asr_malay_datasets = [
32
- 'YouTube ASR: Malay with English Prompt'
33
- ]
34
-
35
-
36
- asr_tamil_datasets = [
37
- 'CommonVoice-17-Tamil',
38
- 'Fleurs-Tamil',
39
- 'YouTube ASR: Tamil with English Prompt'
40
- ]
41
-
42
-
43
- asr_indonesian_datasets = [
44
- 'CommonVoice-17-Indonesian',
45
- 'GigaSpeech-2-Indonesain',
46
- ]
47
-
48
-
49
- asr_thai_datasets = [
50
- 'GigaSpeech-2-Thai',
51
- 'Lotus-Thai'
52
- ]
53
-
54
-
55
- asr_vietnamese_datasets = [
56
- 'CommonVoice-17-Vietnamese',
57
- 'GigaSpeech-2-Vietnamese'
58
- ]
59
-
60
-
61
- asr_private_datasets = [
62
- 'CNA',
63
- 'IDPC',
64
- 'Parliament',
65
- 'UKUS-News',
66
- 'Mediacorp',
67
- 'IDPC-Short',
68
- 'Parliament-Short',
69
- 'UKUS-News-Short',
70
- 'Mediacorp-Short',
71
- 'YouTube ASR: English Singapore Content',
72
- 'YouTube ASR: English with Strong Emotion',
73
- ]
74
-
75
-
76
- speech_translation_datasets = [
77
- 'CoVoST2-EN-ID',
78
- 'CoVoST2-EN-ZH',
79
- 'CoVoST2-EN-TA',
80
- 'CoVoST2-ID-EN',
81
- 'CoVoST2-ZH-EN',
82
- 'CoVoST2-TA-EN'
83
- ]
84
-
85
-
86
- speech_qa_english_datasets = [
87
- 'CN-College-Listen-MCQ',
88
- 'DREAM-TTS-MCQ',
89
- 'SLUE-P2-SQA5',
90
- 'Public-SG-Speech-QA',
91
- 'Spoken-SQuAD',
92
- 'MMAU-mini'
93
- ]
94
-
95
-
96
- speech_qa_singlish_datasets = [
97
- 'MNSC-PART3-SQA',
98
- 'MNSC-PART4-SQA',
99
- 'MNSC-PART5-SQA',
100
- 'MNSC-PART6-SQA',
101
- ]
102
-
103
-
104
- sds_datasets = [
105
- 'MNSC-PART3-SDS',
106
- 'MNSC-PART4-SDS',
107
- 'MNSC-PART5-SDS',
108
- 'MNSC-PART6-SDS',
109
- ]
110
-
111
-
112
- si_datasets = [
113
- 'OpenHermes-Audio',
114
- 'ALPACA-Audio',
115
- ]
116
-
117
-
118
- ac_datasets = [
119
- 'WavCaps',
120
- 'AudioCaps',
121
- ]
122
-
123
-
124
- asqa_datasets = [
125
- 'Clotho-AQA',
126
- 'WavCaps-QA',
127
- 'AudioCaps-QA'
128
- ]
129
-
130
-
131
- er_datasets = [
132
- 'IEMOCAP-Emotion',
133
- 'MELD-Sentiment',
134
- 'MELD-Emotion',
135
- ]
136
-
137
-
138
- ar_datasets = [
139
- 'VoxCeleb-Accent',
140
- 'MNSC-AR-Sentence',
141
- 'MNSC-AR-Dialogue',
142
- ]
143
-
144
-
145
- gr_datasets = [
146
- 'VoxCeleb-Gender',
147
- 'IEMOCAP-Gender'
148
- ]
149
-
150
-
151
- music_datasets = ['MuChoMusic']
152
-
153
-
154
- wer_development_datasets = [
155
- 'YouTube ASR: Malay with Malay Prompt',
156
- 'YouTube ASR: Chinese with Chinese Prompt',
157
- 'SEAME-Dev-Mandarin',
158
- 'SEAME-Dev-Singlish',
159
- ]
160
-
161
-
162
- non_wer_development_datasets = [
163
- 'YouTube SQA: English with Singapore Content',
164
- 'YouTube SDS: English with Singapore Content',
165
- 'YouTube PQA: English with Singapore Content',
166
- ]
167
-
168
-
169
- wer_displayname2datasetname = {
170
- 'LibriSpeech-Clean' : 'librispeech_test_clean',
171
- 'LibriSpeech-Other' : 'librispeech_test_other',
172
- 'CommonVoice-15-EN' : 'common_voice_15_en_test',
173
- 'Peoples-Speech' : 'peoples_speech_test',
174
- 'GigaSpeech-1' : 'gigaspeech_test',
175
- 'Earnings-21' : 'earnings21_test',
176
- 'Earnings-22' : 'earnings22_test',
177
- 'TED-LIUM-3' : 'tedlium3_test',
178
- 'TED-LIUM-3-LongForm' : 'tedlium3_long_form_test',
179
-
180
- 'MNSC-PART1-ASR' : 'imda_part1_asr_test',
181
- 'MNSC-PART2-ASR' : 'imda_part2_asr_test',
182
- 'MNSC-PART3-ASR' : 'imda_part3_30s_asr_test',
183
- 'MNSC-PART4-ASR' : 'imda_part4_30s_asr_test',
184
- 'MNSC-PART5-ASR' : 'imda_part5_30s_asr_test',
185
- 'MNSC-PART6-ASR' : 'imda_part6_30s_asr_test',
186
-
187
- 'AISHELL-ASR-ZH' : 'aishell_asr_zh_test',
188
- 'CommonVoice-ZH' : 'commonvoice_zh_asr',
189
-
190
- 'CommonVoice-17-Indonesian' : 'commonvoice_17_id_asr',
191
- 'CommonVoice-17-Tamil' : 'commonvoice_17_ta_asr',
192
- 'CommonVoice-17-Thai' : 'commonvoice_17_th_asr',
193
- 'CommonVoice-17-Vietnamese' : 'commonvoice_17_vi_asr',
194
- 'GigaSpeech-2-Indonesain' : 'gigaspeech2_id_test',
195
- 'GigaSpeech-2-Thai' : 'gigaspeech2_th_test',
196
- 'GigaSpeech-2-Vietnamese' : 'gigaspeech2_vi_test',
197
- 'Fleurs-Tamil' : 'fleurs_tamil_ta_30_asr',
198
- 'Lotus-Thai' : 'lotus_thai_th_30_asr',
199
-
200
- 'CNA' : 'cna_test',
201
- 'IDPC' : 'idpc_test',
202
- 'Parliament' : 'parliament_test',
203
- 'UKUS-News' : 'ukusnews_test',
204
- 'Mediacorp' : 'mediacorp_test',
205
- 'IDPC-Short' : 'idpc_short_test',
206
- 'Parliament-Short': 'parliament_short_test',
207
- 'UKUS-News-Short' : 'ukusnews_short_test',
208
- 'Mediacorp-Short' : 'mediacorp_short_test',
209
-
210
- 'YouTube ASR: English Singapore Content': 'ytb_asr_batch1',
211
- 'YouTube ASR: English with Strong Emotion': 'ytb_asr_batch2',
212
- 'YouTube ASR: Malay with English Prompt': 'ytb_asr_batch3_malay',
213
- 'YouTube ASR: Chinese with English Prompt': 'ytb_asr_batch3_chinese',
214
- 'YouTube ASR: Tamil with English Prompt': 'ytb_asr_batch3_tamil',
215
- 'YouTube ASR: Tamil with English Prompt V2': 'ytb_asr_batch3_tamil_v2',
216
-
217
- 'YouTube ASR: Malay with Malay Prompt': 'ytb_asr_batch3_ms_ms_prompt',
218
- 'YouTube ASR: Chinese with Chinese Prompt': 'ytb_asr_batch3_zh_zh_prompt',
219
-
220
- 'SEAME-Dev-Mandarin' : 'seame_dev_man',
221
- 'SEAME-Dev-Singlish' : 'seame_dev_sge',
222
- }
223
-
224
-
225
- non_wer_displayname2datasetname = {
226
- 'CoVoST2-EN-ID' : 'covost2_en_id_test',
227
- 'CoVoST2-EN-ZH' : 'covost2_en_zh_test',
228
- 'CoVoST2-EN-TA' : 'covost2_en_ta_test',
229
- 'CoVoST2-ID-EN' : 'covost2_id_en_test',
230
- 'CoVoST2-ZH-EN' : 'covost2_zh_en_test',
231
- 'CoVoST2-TA-EN' : 'covost2_ta_en_test',
232
-
233
- 'CN-College-Listen-MCQ': 'cn_college_listen_mcq_test',
234
- 'DREAM-TTS-MCQ' : 'dream_tts_mcq_test',
235
- 'SLUE-P2-SQA5' : 'slue_p2_sqa5_test',
236
- 'Public-SG-Speech-QA' : 'public_sg_speech_qa_test',
237
- 'Spoken-SQuAD' : 'spoken_squad_test',
238
- 'MMAU-mini' : 'mmau_mini',
239
-
240
- 'MNSC-PART3-SQA' : 'imda_part3_30s_sqa_human_test',
241
- 'MNSC-PART4-SQA' : 'imda_part4_30s_sqa_human_test',
242
- 'MNSC-PART5-SQA' : 'imda_part5_30s_sqa_human_test',
243
- 'MNSC-PART6-SQA' : 'imda_part6_30s_sqa_human_test',
244
-
245
- 'MNSC-PART3-SDS' : 'imda_part3_30s_ds_human_test',
246
- 'MNSC-PART4-SDS' : 'imda_part4_30s_ds_human_test',
247
- 'MNSC-PART5-SDS' : 'imda_part5_30s_ds_human_test',
248
- 'MNSC-PART6-SDS' : 'imda_part6_30s_ds_human_test',
249
-
250
- 'OpenHermes-Audio' : 'openhermes_audio_test',
251
- 'ALPACA-Audio' : 'alpaca_audio_test',
252
-
253
- 'WavCaps' : 'wavcaps_test',
254
- 'AudioCaps' : 'audiocaps_test',
255
-
256
- 'Clotho-AQA' : 'clotho_aqa_test',
257
- 'WavCaps-QA' : 'wavcaps_qa_test',
258
- 'AudioCaps-QA' : 'audiocaps_qa_test',
259
-
260
- 'IEMOCAP-Emotion' : 'iemocap_emotion_test',
261
- 'MELD-Sentiment' : 'meld_sentiment_test',
262
- 'MELD-Emotion' : 'meld_emotion_test',
263
-
264
- 'VoxCeleb-Accent' : 'voxceleb_accent_test',
265
- 'MNSC-AR-Sentence' : 'imda_ar_sentence',
266
- 'MNSC-AR-Dialogue' : 'imda_ar_dialogue',
267
-
268
- 'VoxCeleb-Gender' : 'voxceleb_gender_test',
269
- 'IEMOCAP-Gender' : 'iemocap_gender_test',
270
-
271
- 'MuChoMusic' : 'muchomusic_test',
272
-
273
- 'YouTube SQA: English with Singapore Content': 'ytb_sqa_batch1',
274
- 'YouTube SDS: English with Singapore Content': 'ytb_sds_batch1',
275
- 'YouTube PQA: English with Singapore Content': 'ytb_pqa_batch1',
276
-
277
- 'YouTube SQA: Malay': 'ytb_sqa_batch3_malay',
278
- 'YouTube SQA: Chinese': 'ytb_sqa_batch3_chinese',
279
- 'YouTube SQA: Tamil': 'ytb_sqa_batch3_tamil',
280
-
281
- 'YouTube SDS: Malay': 'ytb_sds_batch3_malay',
282
- 'YouTube SDS: Chinese': 'ytb_sds_batch3_chinese',
283
- 'YouTube SDS: Tamil': 'ytb_sds_batch3_tamil',
284
- }
285
-
286
-
287
- displayname2datasetname = {}
288
- displayname2datasetname.update(wer_displayname2datasetname)
289
- displayname2datasetname.update(non_wer_displayname2datasetname)
290
-
291
-
292
- datasetname2diaplayname = {datasetname: displayname for displayname, datasetname in displayname2datasetname.items()}
293
-
294
-
295
- dataset_diaplay_information = {
296
- 'LibriSpeech-Clean' : 'A clean, high-quality testset of the LibriSpeech dataset, used for ASR testing.',
297
- 'LibriSpeech-Other' : 'A more challenging, noisier testset of the LibriSpeech dataset for ASR testing.',
298
- 'CommonVoice-15-EN' : 'Test set from the Common Voice project, which is a crowd-sourced, multilingual speech dataset.',
299
- 'Peoples-Speech' : 'A large-scale, open-source speech recognition dataset, with diverse accents and domains.',
300
- 'GigaSpeech-1' : 'A large-scale ASR dataset with diverse audio sources like podcasts, interviews, etc.',
301
- 'Earnings-21' : 'ASR test dataset focused on earnings calls from 2021, with professional speech and financial jargon.',
302
- 'Earnings-22' : 'Similar to Earnings21, but covering earnings calls from 2022.',
303
- 'TED-LIUM-3' : 'A test set derived from TED talks, covering diverse speakers and topics.',
304
- 'TED-LIUM-3-LongForm' : 'A longer version of the TED-LIUM dataset, containing extended audio samples. This poses challenges to existing fusion methods in handling long audios. However, it provides benchmark for future development.',
305
- 'AISHELL-ASR-ZH' : 'ASR test dataset for Mandarin Chinese, based on the Aishell dataset.',
306
- 'CoVoST2-EN-ID' : 'CoVoST 2 dataset for speech translation from English to Indonesian.',
307
- 'CoVoST2-EN-ZH' : 'CoVoST 2 dataset for speech translation from English to Chinese.',
308
- 'CoVoST2-EN-TA' : 'CoVoST 2 dataset for speech translation from English to Tamil.',
309
- 'CoVoST2-ID-EN' : 'CoVoST 2 dataset for speech translation from Indonesian to English.',
310
- 'CoVoST2-ZH-EN' : 'CoVoST 2 dataset for speech translation from Chinese to English.',
311
- 'CoVoST2-TA-EN' : 'CoVoST 2 dataset for speech translation from Tamil to English.',
312
- 'CN-College-Listen-MCQ': 'Chinese College English Listening Test, with multiple-choice questions.',
313
- 'DREAM-TTS-MCQ' : 'DREAM dataset for spoken question-answering, derived from textual data and synthesized speech.',
314
- 'SLUE-P2-SQA5' : 'Spoken Language Understanding Evaluation (SLUE) dataset, part 2, focused on QA tasks.',
315
- 'Public-SG-Speech-QA' : 'Public dataset for speech-based question answering, gathered from Singapore.',
316
- 'Spoken-SQuAD' : 'Spoken SQuAD dataset, based on the textual SQuAD dataset, converted into audio.',
317
- 'OpenHermes-Audio' : 'Test set for spoken instructions. Synthesized from the OpenHermes dataset.',
318
- 'ALPACA-Audio' : 'Spoken version of the ALPACA dataset, used for evaluating instruction following in audio.',
319
- 'WavCaps' : 'WavCaps is a dataset for testing audio captioning, where models generate textual descriptions of audio clips.',
320
- 'AudioCaps' : 'AudioCaps dataset, used for generating captions from general audio events.',
321
- 'Clotho-AQA' : 'Clotho dataset adapted for audio-based question answering, containing audio clips and questions.',
322
- 'WavCaps-QA' : 'Question-answering test dataset derived from WavCaps, focusing on audio content.',
323
- 'AudioCaps-QA' : 'AudioCaps adapted for question-answering tasks, using audio events as input for Q&A.',
324
- 'VoxCeleb-Accent' : 'Test dataset for accent recognition, based on VoxCeleb, a large speaker identification dataset.',
325
- 'MNSC-AR-Sentence' : 'Accent recognition based on the IMDA NSC dataset, focusing on sentence-level accents.',
326
- 'MNSC-AR-Dialogue' : 'Accent recognition based on the IMDA NSC dataset, focusing on dialogue-level accents.',
327
-
328
- 'VoxCeleb-Gender': 'Test dataset for gender classification, also derived from VoxCeleb.',
329
- 'IEMOCAP-Gender' : 'Gender classification based on the IEMOCAP dataset.',
330
- 'IEMOCAP-Emotion': 'Emotion recognition test data from the IEMOCAP dataset, focusing on identifying emotions in speech.',
331
- 'MELD-Sentiment' : 'Sentiment recognition from speech using the MELD dataset, classifying positive, negative, or neutral sentiments.',
332
- 'MELD-Emotion' : 'Emotion classification in speech using MELD, detecting specific emotions like happiness, anger, etc.',
333
- 'MuChoMusic' : 'Test dataset for music understanding, from paper: MuChoMusic: Evaluating Music Understanding in Multimodal Audio-Language Models.',
334
- 'MNSC-PART1-ASR' : 'Speech recognition test data from the IMDA NSC project, Part 1.',
335
- 'MNSC-PART2-ASR' : 'Speech recognition test data from the IMDA NSC project, Part 2.',
336
- 'MNSC-PART3-ASR' : 'Speech recognition test data from the IMDA NSC project, Part 3.',
337
- 'MNSC-PART4-ASR' : 'Speech recognition test data from the IMDA NSC project, Part 4.',
338
- 'MNSC-PART5-ASR' : 'Speech recognition test data from the IMDA NSC project, Part 5.',
339
- 'MNSC-PART6-ASR' : 'Speech recognition test data from the IMDA NSC project, Part 6.',
340
- 'MNSC-PART3-SQA' : 'Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 3.',
341
- 'MNSC-PART4-SQA' : 'Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 4.',
342
- 'MNSC-PART5-SQA' : 'Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 5.',
343
- 'MNSC-PART6-SQA' : 'Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 6.',
344
- 'MNSC-PART3-SDS' : 'Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 3.',
345
- 'MNSC-PART4-SDS' : 'Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 4.',
346
- 'MNSC-PART5-SDS' : 'Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 5.',
347
- 'MNSC-PART6-SDS' : 'Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 6.',
348
-
349
- 'CNA' : 'Under Development',
350
- 'IDPC' : 'Under Development',
351
- 'Parliament' : 'Under Development',
352
- 'UKUS-News' : 'Under Development',
353
- 'Mediacorp' : 'Under Development',
354
- 'IDPC-Short' : 'Under Development',
355
- 'Parliament-Short': 'Under Development',
356
- 'UKUS-News-Short' : 'Under Development',
357
- 'Mediacorp-Short' : 'Under Development',
358
-
359
- 'CommonVoice-ZH' : 'Under Development',
360
- 'CommonVoice-17-Indonesian' : 'Under Development',
361
- 'CommonVoice-17-Tamil' : 'Under Development',
362
- 'CommonVoice-17-Thai' : 'Under Development',
363
- 'CommonVoice-17-Vietnamese' : 'Under Development',
364
- 'GigaSpeech-2-Indonesain' : 'Under Development',
365
- 'GigaSpeech-2-Thai' : 'Under Development',
366
- 'GigaSpeech-2-Vietnamese' : 'Under Development',
367
- 'Fleurs-Tamil' : 'Under Development',
368
- 'Lotus-Thai' : 'Under Development',
369
- 'MMAU-mini' : 'Under Development',
370
-
371
-
372
- 'YouTube ASR: English Singapore Content' : 'YouTube Evaluation Dataset for ASR Task: <br> This dataset contains English and Singlish audio clips, featuring Singapore-related content. <br> It includes approximately 2.5 hours of audio, with individual clips ranging from 2 seconds to 30 seconds in length.',
373
-
374
- 'YouTube ASR: English with Strong Emotion' : 'YouTube Evaluation Dataset for ASR Task: <br> This dataset contains English, Singlish and some unknown languages audio clips, featuring speech with strong emotional expression. <br> It includes approximately 3.9 hours of audio, with each clip lasting 30 seconds.',
375
-
376
- 'YouTube ASR: Malay with English Prompt': 'YouTube Evaluation Dataset for ASR Task: <br> This dataset mainly contains Malay and some Malay-English codeswitch audio clips, featuring with English prompts. <br> It includes approximately 2.55 hours of audio, with indicidual clips ranging form 30 seconds to 95 seconds in length.',
377
-
378
- 'YouTube ASR: Malay with Malay Prompt': 'YouTube Evaluation Dataset for ASR Task: <br> This dataset use the same audio from <i>YouTube ASR: Malay English Prompt</i>, except featuring with Malay prompts. <br> It includes approximately 2.55 hours of audio, with indicidual clips ranging form 30 seconds to 95 seconds in length.',
379
-
380
- 'YouTube ASR: Chinese with English Prompt': 'YouTube Evaluation Dataset for ASR Task: <br> This dataset contains Chinese and some Chinese-English codeswitch audio clips, featuring with English prompts. <br> It includes approximately 3.32 hours of audio, with individual clips ranging from 17 seconds to 1966 seconds in length.',
381
-
382
- 'YouTube ASR: Chinese with Chinese Prompt': 'YouTube Evaluation Dataset for ASR Task: <br> This dataset contains Chinese and some Chinese-English codeswitch audio clips, featuring with Chinese prompts. <br> It includes approximately 3.32 hours of audio, with individual clips ranging from 17 seconds to 1966 seconds in length.',
383
-
384
- 'YouTube ASR: Tamil with Tamil Prompt': 'YouTube Evaluation Dataset for ASR Task: <br> This dataset contains Tamil and some Tamil-English codeswitch audio clips, featuring with Tamil prompts. <br> It includes approximately 2.44 hours of audio, with individual clips ranging from 30 seconds to 324 seconds in length.',
385
-
386
- 'YouTube ASR: Tamil with English Prompt': 'YouTube Evaluation Dataset for ASR Task: <br> This dataset contains Tamil and some Tamil-English codeswitch audio clips, featuring with English prompts. <br> It includes approximately 2.44 hours of audio, with individual clips ranging from 30 seconds to 324 seconds in length.',
387
-
388
- 'YouTube ASR: Tamil with English Prompt V2': 'YouTube Evaluation Dataset for ASR Task: <br> This dataset contains Tamil and some Tamil-English codeswitch audio clips, featuring with English prompts. <br> It includes approximately 2.44 hours of audio, with individual clips ranging from 30 seconds to 324 seconds in length.',
389
-
390
- 'YouTube ASR Translation: Malay2English': 'YouTube Evaluation Dataset for ASR Task: <br> The audio of dataset is same as <i>YouTube ASR: Malay<i>',
391
-
392
- # 'YouTube ASR Translation: Chinese2English': 'YouTube Evaluation Dataset for ASR Task: <br> The audio of dataset is same as <i>YouTube ASR: Chinese<i>',
393
-
394
- # 'YouTube ASR Translation: Tamil2English': 'YouTube Evaluation Dataset for ASR Task: <br> The audio of dataset is same as <i>YouTube ASR: Tamil<i>',
395
-
396
-
397
-
398
- 'SEAME-Dev-Mandarin' : 'Under Development',
399
- 'SEAME-Dev-Singlish' : 'Under Development',
400
-
401
- 'YouTube SQA: English with Singapore Content': 'YouTube Evaluation Dataset for Speech-QA Task: <br> This dataset contains English and Singlish audio clips, featuring Singapore-related content. <br> It includes approximately 7.6 hours of audio, with individual clips ranging from 8 seconds to 32 seconds in length.',
402
-
403
- 'YouTube SQA: Malay': 'YouTube Evaluation Dataset for Speech-QA Task: <br> The auido of this dataset is same as <i>YouTube ASR: Malay<i>, it contains Malay and some Malay-English codeswitch audio clips, featuring with English prompts. <br> It includes approximately 2.55 hours of audio, with indicidual clips ranging form 30 seconds to 95 seconds in length.',
404
-
405
- 'YouTube SQA: Chinese': 'YouTube Evaluation Dataset for Speech-QA Task: <br> The auido of this dataset is same as <i>YouTube ASR: Chinese<i>',
406
-
407
- 'YouTube SQA: Tamil': 'YouTube Evaluation Dataset for Speech-QA Task: <br> The auido of this dataset is same as <i>YouTube ASR: Tamil<i>',
408
-
409
- 'YouTube SDS: English with Singapore Content': 'YouTube Evaluation Dataset for Summary Task: <br> This dataset contains English and Singlish audio clips, featuring Singapore-related content. <br> It includes approximately 5.4 hours of audio, with individual clips ranging from 8 seconds to 32 seconds in length.',
410
-
411
- 'YouTube SDS: Malay': 'YouTube Evaluation Dataset for Speech-QA Task: <br> The auido of this dataset is same as <i>YouTube ASR: Malay<i>, it contains Malay and some Malay-English codeswitch audio clips, featuring with English prompts. <br> It includes approximately 2.55 hours of audio, with indicidual clips ranging form 30 seconds to 95 seconds in length.',
412
-
413
- 'YouTube SDS: Chinese': 'YouTube Evaluation Dataset for Speech-QA Task: <br> The auido of this dataset is same as <i>YouTube ASR: Chinese<i>',
414
-
415
- 'YouTube SDS: Tamil': 'YouTube Evaluation Dataset for Speech-QA Task: <br> The auido of this dataset is same as <i>YouTube ASR: Tamil<i>',
416
-
417
-
418
- 'YouTube PQA: English with Singapore Content': 'YouTube Evaluation Dataset for Paralinguistics QA Task: <br> This dataset contains English and Singlish audio clips, featuring Singapore-related content. <br> It includes approximately 41.4 hours of audio, with individual clips ranging from 41 seconds to 83 seconds in length.',
419
-
420
-
421
- }
422
-
423
-
424
- metrics_info = {
425
- 'wer' : 'Word Error Rate (WER) - The Lower, the better.',
426
- 'llama3_70b_judge_binary': 'Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.',
427
- 'llama3_70b_judge' : 'Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.',
428
- 'meteor' : 'METEOR Score. The higher, the better.',
429
- 'bleu' : 'BLEU Score. The higher, the better.',
430
- }
431
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
app/draw_diagram.py DELETED
@@ -1,193 +0,0 @@
1
- import streamlit as st
2
- import pandas as pd
3
- import numpy as np
4
- from streamlit_echarts import st_echarts
5
- from app.show_examples import *
6
- from app.content import *
7
-
8
- import pandas as pd
9
-
10
- from app.content import wer_displayname2datasetname
11
- from model_information import get_dataframe
12
- info_df = get_dataframe()
13
-
14
-
15
- def draw(folder_name, category_name, displayname, metrics, cus_sort=True):
16
-
17
- folder = f"./results_organized/{metrics}/"
18
-
19
- # Load the results from CSV
20
- data_path = f'{folder}/{category_name.lower()}.csv'
21
- chart_data = pd.read_csv(data_path).round(3)
22
-
23
- dataset_name = displayname2datasetname[displayname]
24
- chart_data = chart_data[['Model', dataset_name]]
25
-
26
- # Rename to proper display name
27
- chart_data = chart_data.rename(columns=datasetname2diaplayname)
28
-
29
- st.markdown("""
30
- <style>
31
- .stMultiSelect [data-baseweb=select] span {
32
- max-width: 800px;
33
- font-size: 0.9rem;
34
- background-color: #3C6478 !important; /* Background color for selected items */
35
- color: white; /* Change text color */
36
- back
37
- }
38
- </style>
39
- """, unsafe_allow_html=True)
40
-
41
- # remap model names
42
- display_model_names = {key.strip() :val.strip() for key, val in zip(info_df['Original Name'], info_df['Proper Display Name'])}
43
- chart_data['model_show'] = chart_data['Model'].map(lambda x: display_model_names.get(x, x))
44
-
45
-
46
- models = st.multiselect("Please choose the model",
47
- sorted(chart_data['model_show'].tolist()),
48
- default = sorted(chart_data['model_show'].tolist()),
49
- )
50
-
51
- chart_data = chart_data[chart_data['model_show'].isin(models)]
52
- chart_data = chart_data.sort_values(by=[displayname], ascending=cus_sort).dropna(axis=0)
53
-
54
- if len(chart_data) == 0: return
55
-
56
-
57
- # = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = =
58
- '''
59
- Show Table
60
- '''
61
- with st.container():
62
- st.markdown('##### TABLE')
63
-
64
-
65
- model_link = {key.strip(): val for key, val in zip(info_df['Proper Display Name'], info_df['Link'])}
66
-
67
- chart_data['model_link'] = chart_data['model_show'].map(model_link)
68
-
69
- chart_data_table = chart_data[['model_show', chart_data.columns[1], chart_data.columns[3]]]
70
-
71
- # Format numeric columns to 2 decimal places
72
- #chart_data_table[chart_data_table.columns[1]] = chart_data_table[chart_data_table.columns[1]].apply(lambda x: round(float(x), 3) if isinstance(float(x), (int, float)) else float(x))
73
- cur_dataset_name = chart_data_table.columns[1]
74
-
75
-
76
- def highlight_first_element(x):
77
- # Create a DataFrame with the same shape as the input
78
- df_style = pd.DataFrame('', index=x.index, columns=x.columns)
79
- # Apply background color to the first element in row 0 (df[0][0])
80
- # df_style.iloc[0, 1] = 'background-color: #b0c1d7; color: white'
81
- df_style.iloc[0, 1] = 'background-color: #b0c1d7'
82
-
83
- return df_style
84
-
85
- if cur_dataset_name in wer_displayname2datasetname:
86
- chart_data_table = chart_data_table.sort_values(
87
- by=chart_data_table.columns[1],
88
- ascending=True
89
- ).reset_index(drop=True)
90
- else:
91
- chart_data_table = chart_data_table.sort_values(
92
- by=chart_data_table.columns[1],
93
- ascending=False
94
- ).reset_index(drop=True)
95
-
96
-
97
- styled_df = chart_data_table.style.format(
98
- {chart_data_table.columns[1]: "{:.3f}"}
99
- ).apply(
100
- highlight_first_element, axis=None
101
- )
102
-
103
-
104
- st.dataframe(
105
- styled_df,
106
- column_config={
107
- 'model_show': 'Model',
108
- chart_data_table.columns[1]: {'alignment': 'left'},
109
- "model_link": st.column_config.LinkColumn(
110
- "Model Link",
111
- ),
112
- },
113
- hide_index=True,
114
- use_container_width=True
115
- )
116
-
117
-
118
- # = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = =
119
- '''
120
- Show Chart
121
- '''
122
-
123
- # Initialize a session state variable for toggling the chart visibility
124
- if "show_chart" not in st.session_state:
125
- st.session_state.show_chart = False
126
-
127
- # Create a button to toggle visibility
128
- if st.button("Show Chart"):
129
- st.session_state.show_chart = not st.session_state.show_chart
130
-
131
- if st.session_state.show_chart:
132
-
133
- with st.container():
134
- st.markdown('##### CHART')
135
-
136
- # Get Values
137
- data_values = chart_data.iloc[:, 1]
138
-
139
- # Calculate Q1 and Q3
140
- q1 = data_values.quantile(0.25)
141
- q3 = data_values.quantile(0.75)
142
-
143
- # Calculate IQR
144
- iqr = q3 - q1
145
-
146
- # Define lower and upper bounds (1.5*IQR is a common threshold)
147
- lower_bound = q1 - 1.5 * iqr
148
- upper_bound = q3 + 1.5 * iqr
149
-
150
- # Filter data within the bounds
151
- filtered_data = data_values[(data_values >= lower_bound) & (data_values <= upper_bound)]
152
-
153
- # Calculate min and max values after outlier handling
154
- min_value = round(filtered_data.min() - 0.1 * filtered_data.min(), 3)
155
- max_value = round(filtered_data.max() + 0.1 * filtered_data.max(), 3)
156
-
157
- options = {
158
- # "title": {"text": f"{dataset_name}"},
159
- "tooltip": {
160
- "trigger": "axis",
161
- "axisPointer": {"type": "cross", "label": {"backgroundColor": "#6a7985"}},
162
- "triggerOn": 'mousemove',
163
- },
164
- "legend": {"data": ['Overall Accuracy']},
165
- "toolbox": {"feature": {"saveAsImage": {}}},
166
- "grid": {"left": "3%", "right": "4%", "bottom": "3%", "containLabel": True},
167
- "xAxis": [
168
- {
169
- "type": "category",
170
- "boundaryGap": True,
171
- "triggerEvent": True,
172
- "data": chart_data['model_show'].tolist(),
173
- }
174
- ],
175
- "yAxis": [{"type": "value",
176
- "min": min_value,
177
- "max": max_value,
178
- "boundaryGap": True
179
- # "splitNumber": 10
180
- }],
181
- "series": [{
182
- "name": f"{dataset_name}",
183
- "type": "bar",
184
- "data": chart_data[f'{displayname}'].tolist(),
185
- }],
186
- }
187
-
188
- events = {
189
- "click": "function(params) { return params.value }"
190
- }
191
-
192
- value = st_echarts(options=options, events=events, height="500px")
193
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
app/pages.py DELETED
@@ -1,576 +0,0 @@
1
- import streamlit as st
2
- from app.draw_diagram import *
3
- from app.content import *
4
- from app.summarization import *
5
- from app.show_examples import *
6
-
7
- def dataset_contents(dataset, metrics):
8
-
9
- custom_css = """
10
- <style>
11
- .my-dataset-info {
12
- # background-color: #F9EBEA;
13
- # padding: 10px;
14
- color: #050505;
15
- font-style: normal;
16
- font-size: 8px;
17
- height: auto;
18
- }
19
- </style>
20
- """
21
- st.markdown(custom_css, unsafe_allow_html=True)
22
- st.markdown(f"""<div class="my-dataset-info">
23
- <p><b>About this dataset</b>:</p>
24
- <p>{dataset}</p>
25
- </div>""", unsafe_allow_html=True)
26
- st.markdown(f"""<div class="my-dataset-info">
27
- <p><b>About this metric</b>:</p>
28
- <p>{metrics}</p>
29
- </div>""", unsafe_allow_html=True)
30
-
31
-
32
- def dashboard():
33
-
34
- with st.container():
35
- st.title("Leaderboard for AudioBench")
36
-
37
- st.markdown("""
38
- [gh1]: https://github.com/AudioLLMs/AudioBench
39
- [gh2]: https://github.com/AudioLLMs/AudioBench
40
- **Toolkit:** [![GitHub Repo stars](https://img.shields.io/github/stars/AudioLLMs/AudioBench?style=social)][gh1] |
41
- [**Paper @ NAACL 2025**](https://arxiv.org/abs/2406.16020) |
42
- **Resource for AudioLLMs:** [![GitHub Repo stars](https://img.shields.io/github/stars/AudioLLMs/Awesome-Audio-LLM?style=social)][gh2]
43
- """)
44
-
45
-
46
- st.markdown("""
47
- #### Recent updates
48
- - **Nov. 2025**: Added Omnilingual-LLM-ASR-7B and MERaLiON-SpeechEncoder2-ASR-CTC to the leaderboard.
49
- - **May. 2025**: Further expanded ASR task to 7 different languages. Added MERaLiON-2 to the leaderboard.
50
- - **Jan. 2025**: AudioBench is officially accepted to NAACL 2025!
51
- - **Jan. 2025**: Update the layout.
52
- - **Dec. 2024**: Added MuChoMusic dataset for Music Understanding - MCQ Questions. From Paper: https://arxiv.org/abs/2408.01337.
53
- - **Dec. 2024**: Singlish ASR task added! The datasets are available on [HF](https://huggingface.co/datasets/MERaLiON/MNSC).
54
- - **Dec. 2024**: Updated layout and added support for comparison between models with similar sizes. 1) Reorganized layout for a better user experience. 2) Added performance summary for each task.
55
- - **Aug. 2024**: Initial leaderboard is now online.
56
- """)
57
-
58
- st.divider()
59
-
60
- st.markdown("""
61
- #### Evaluating Audio-based Large Language Models
62
-
63
- - AudioBench is a comprehensive evaluation benchmark designed for general instruction-following audio large language models.
64
- - AudioBench is an evaluation benchmark that we continually improve and maintain.
65
-
66
- Below are the initial 26 datasets that are included in AudioBench. We are now exteneded to over 40 datasets and going to extend to more in the future.
67
- """
68
- )
69
-
70
-
71
- with st.container():
72
-
73
- st.markdown('''
74
- ''')
75
-
76
- st.markdown("###### :dart: Our Benchmark includes: ")
77
- cols = st.columns(8)
78
- cols[0].metric(label="Tasks", value=">8")
79
- cols[1].metric(label="Datasets", value=">40")
80
- cols[2].metric(label="Evaluated Models", value=">5")
81
-
82
- st.divider()
83
- with st.container():
84
- left_co, right_co = st.columns([1, 0.1])
85
-
86
- with left_co:
87
- st.markdown("""
88
- ##### Citations :round_pushpin:
89
- ```
90
- @article{wang2024audiobench,
91
- title={AudioBench: A Universal Benchmark for Audio Large Language Models},
92
- author={Wang, Bin and Zou, Xunlong and Lin, Geyu and Sun, Shuo and Liu, Zhuohan and Zhang, Wenyu and Liu, Zhengyuan and Aw, AiTi and Chen, Nancy F},
93
- journal={NAACL},
94
- year={2025}
95
- }
96
- ```
97
- ```
98
- @article{zhang2024mowe,
99
- title={MoWE-Audio: Multitask AudioLLMs with Mixture of Weak Encoders},
100
- author={Zhang, Wenyu and Sun, Shuo and Wang, Bin and Zou, Xunlong and Liu, Zhuohan and He, Yingxu and Lin, Geyu and Chen, Nancy F and Aw, Ai Ti},
101
- journal={ICASSP},
102
- year={2025}
103
- }
104
- ```
105
- ```
106
- @article{wang2025advancing,
107
- title={Advancing Singlish Understanding: Bridging the Gap with Datasets and Multimodal Models},
108
- author={Wang, Bin and Zou, Xunlong and Sun, Shuo and Zhang, Wenyu and He, Yingxu and Liu, Zhuohan and Wei, Chengwei and Chen, Nancy F and Aw, AiTi},
109
- journal={arXiv preprint arXiv:2501.01034},
110
- year={2025}
111
- }
112
- ```
113
- ```
114
- @article{he2024meralion,
115
- title={MERaLiON-AudioLLM: Technical Report},
116
- author={He, Yingxu and Liu, Zhuohan and Sun, Shuo and Wang, Bin and Zhang, Wenyu and Zou, Xunlong and Chen, Nancy F and Aw, Ai Ti},
117
- journal={arXiv preprint arXiv:2412.09818},
118
- year={2024}
119
- }
120
- ```
121
-
122
- """)
123
-
124
-
125
- def asr_english():
126
- st.title("Task: Automatic Speech Recognition - English")
127
-
128
- sum = ['Overall']
129
-
130
- filters_levelone = sum + asr_english_datasets
131
-
132
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
133
-
134
- with left:
135
- filter_1 = st.selectbox('Dataset', filters_levelone)
136
-
137
- if filter_1:
138
- if filter_1 in sum:
139
- sum_table_mulit_metrix('asr_english', ['wer'])
140
- else:
141
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['wer'])
142
- draw('su', 'asr_english', filter_1, 'wer', cus_sort=True)
143
-
144
-
145
- def asr_singlish():
146
- st.title("Task: Automatic Speech Recognition - Singlish")
147
-
148
- sum = ['Overall']
149
-
150
- filters_levelone = sum + asr_singlish_datasets
151
-
152
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
153
-
154
- with left:
155
- filter_1 = st.selectbox('Dataset', filters_levelone)
156
-
157
- if filter_1:
158
- if filter_1 in sum:
159
- sum_table_mulit_metrix('asr_singlish', ['wer'])
160
- else:
161
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['wer'])
162
- draw('su', 'asr_singlish', filter_1, 'wer')
163
-
164
-
165
- def asr_mandarin():
166
- st.title("Task: Automatic Speech Recognition - Mandarin")
167
-
168
- sum = ['Overall']
169
-
170
- filters_levelone = sum + asr_mandarin_datasets
171
-
172
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
173
-
174
- with left:
175
- filter_1 = st.selectbox('Dataset', filters_levelone)
176
-
177
- if filter_1:
178
- if filter_1 in sum:
179
- sum_table_mulit_metrix('asr_mandarin', ['wer'])
180
- else:
181
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['wer'])
182
- draw('su', 'asr_mandarin', filter_1, 'wer')
183
-
184
-
185
- def asr_malay():
186
- st.title("Task: Automatic Speech Recognition - Malay")
187
-
188
- sum = ['Overall']
189
-
190
- filters_levelone = sum + asr_malay_datasets
191
-
192
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
193
-
194
- with left:
195
- filter_1 = st.selectbox('Dataset', filters_levelone)
196
-
197
- if filter_1:
198
- if filter_1 in sum:
199
- sum_table_mulit_metrix('asr_malay', ['wer'])
200
- else:
201
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['wer'])
202
- draw('su', 'asr_malay', filter_1, 'wer')
203
-
204
-
205
- def asr_tamil():
206
- st.title("Task: Automatic Speech Recognition - Tamil")
207
-
208
- sum = ['Overall']
209
-
210
- filters_levelone = sum + asr_tamil_datasets
211
-
212
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
213
-
214
- with left:
215
- filter_1 = st.selectbox('Dataset', filters_levelone)
216
-
217
- if filter_1:
218
- if filter_1 in sum:
219
- sum_table_mulit_metrix('asr_tamil', ['wer'])
220
- else:
221
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['wer'])
222
- draw('su', 'asr_tamil', filter_1, 'wer')
223
-
224
-
225
- def asr_indonesian():
226
- st.title("Task: Automatic Speech Recognition - Indonesian")
227
-
228
- sum = ['Overall']
229
-
230
- filters_levelone = sum + asr_indonesian_datasets
231
-
232
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
233
-
234
- with left:
235
- filter_1 = st.selectbox('Dataset', filters_levelone)
236
-
237
- if filter_1:
238
- if filter_1 in sum:
239
- sum_table_mulit_metrix('asr_indonesian', ['wer'])
240
- else:
241
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['wer'])
242
- draw('su', 'asr_indonesian', filter_1, 'wer')
243
-
244
-
245
- def asr_thai():
246
- st.title("Task: Automatic Speech Recognition - Thai")
247
-
248
- sum = ['Overall']
249
-
250
- filters_levelone = sum + asr_thai_datasets
251
-
252
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
253
-
254
- with left:
255
- filter_1 = st.selectbox('Dataset', filters_levelone)
256
-
257
- if filter_1:
258
- if filter_1 in sum:
259
- sum_table_mulit_metrix('asr_thai', ['wer'])
260
- else:
261
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['wer'])
262
- draw('su', 'asr_thai', filter_1, 'wer')
263
-
264
-
265
- def asr_vietnamese():
266
- st.title("Task: Automatic Speech Recognition - Vietnamese")
267
-
268
- sum = ['Overall']
269
-
270
- filters_levelone = sum + asr_vietnamese_datasets
271
-
272
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
273
-
274
- with left:
275
- filter_1 = st.selectbox('Dataset', filters_levelone)
276
-
277
- if filter_1:
278
- if filter_1 in sum:
279
- sum_table_mulit_metrix('asr_vietnamese', ['wer'])
280
- else:
281
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['wer'])
282
- draw('su', 'asr_vietnamese', filter_1, 'wer')
283
-
284
-
285
- def asr_private():
286
- st.title("Task: Automatic Speech Recognition - Other Datasets")
287
-
288
- sum = ['Overall']
289
-
290
- filters_levelone = sum + asr_private_datasets
291
-
292
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
293
-
294
- with left:
295
- filter_1 = st.selectbox('Dataset', filters_levelone)
296
-
297
- if filter_1:
298
- if filter_1 in sum:
299
- sum_table_mulit_metrix('asr_private', ['wer'])
300
- else:
301
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['wer'])
302
- draw('su', 'asr_private', filter_1, 'wer')
303
-
304
-
305
- def speech_translation():
306
- st.title("Task: Speech Translation")
307
-
308
- sum = ['Overall']
309
-
310
- filters_levelone = sum + speech_translation_datasets
311
-
312
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
313
-
314
- with left:
315
- filter_1 = st.selectbox('Dataset', filters_levelone)
316
-
317
- if filter_1:
318
- if filter_1 in sum:
319
- sum_table_mulit_metrix('st', ['bleu'])
320
- else:
321
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['bleu'])
322
- draw('su', 'ST', filter_1, 'bleu')
323
-
324
-
325
- def speech_question_answering_english():
326
- st.title("Task: Spoken Question Answering - English")
327
-
328
- sum = ['Overall']
329
-
330
- filters_levelone = sum + speech_qa_english_datasets
331
-
332
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
333
-
334
- with left:
335
- filter_1 = st.selectbox('Dataset', filters_levelone)
336
-
337
- if filter_1:
338
- if filter_1 in sum:
339
- sum_table_mulit_metrix('sqa_english', ['llama3_70b_judge'])
340
-
341
- #elif filter_1 in dataset_lists:
342
- # dataset_contents(sqa_datasets[filter_1], metrics['llama3_70b_judge'])
343
- # draw('su', 'SQA', filter_1, 'llama3_70b_judge')
344
-
345
- else:
346
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['llama3_70b_judge'])
347
- draw('su', 'sqa_english', filter_1, 'llama3_70b_judge')
348
-
349
-
350
- def speech_question_answering_singlish():
351
- st.title("Task: Spoken Question Answering - Singlish")
352
-
353
- sum = ['Overall']
354
-
355
- filters_levelone = sum + speech_qa_singlish_datasets
356
-
357
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
358
-
359
- with left:
360
- filter_1 = st.selectbox('Dataset', filters_levelone)
361
-
362
- if filter_1:
363
- if filter_1 in sum:
364
- sum_table_mulit_metrix('sqa_singlish', ['llama3_70b_judge'])
365
-
366
- else:
367
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['llama3_70b_judge'])
368
- draw('su', 'sqa_singlish', filter_1, 'llama3_70b_judge')
369
-
370
-
371
- def spoken_dialogue_summarization_singlish():
372
- st.title("Task: Spoken Dialogue Summarization - Singlish")
373
-
374
- sum = ['Overall']
375
-
376
- filters_levelone = sum + sds_datasets
377
-
378
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
379
-
380
- with left:
381
- filter_1 = st.selectbox('Dataset', filters_levelone)
382
-
383
- if filter_1:
384
- if filter_1 in sum:
385
- sum_table_mulit_metrix('sds_singlish', ['llama3_70b_judge'])
386
-
387
- else:
388
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['llama3_70b_judge'])
389
- draw('su', 'sds_singlish', filter_1, 'llama3_70b_judge')
390
-
391
-
392
- def speech_instruction():
393
- st.title("Task: Speech Instruction")
394
-
395
- sum = ['Overall']
396
-
397
- filters_levelone = sum + si_datasets
398
-
399
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
400
-
401
- with left:
402
- filter_1 = st.selectbox('Dataset', filters_levelone)
403
-
404
- if filter_1:
405
- if filter_1 in sum:
406
- sum_table_mulit_metrix('speech_instruction', ['llama3_70b_judge'])
407
- else:
408
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['llama3_70b_judge'])
409
- draw('su', 'speech_instruction', filter_1, 'llama3_70b_judge')
410
-
411
-
412
- def audio_captioning():
413
- st.title("Task: Audio Captioning")
414
-
415
- filters_levelone = ac_datasets
416
-
417
- filters_leveltwo = ['Llama3-70b-judge', 'Meteor']
418
-
419
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
420
-
421
- with left:
422
- filter_1 = st.selectbox('Dataset', filters_levelone)
423
- with middle:
424
- metric = st.selectbox('Metric', filters_leveltwo)
425
-
426
- if filter_1 or metric:
427
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info[metric.lower().replace('-', '_')])
428
- draw('asu', 'audio_captioning', filter_1, metric.lower().replace('-', '_'))
429
-
430
-
431
- def audio_scene_question_answering():
432
- st.title("Task: Audio Scene Question Answering")
433
-
434
- sum = ['Overall']
435
-
436
- filters_levelone = sum + asqa_datasets
437
-
438
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
439
-
440
- with left:
441
- filter_1 = st.selectbox('Dataset', filters_levelone)
442
-
443
- if filter_1:
444
- if filter_1 in sum:
445
- sum_table_mulit_metrix('audio_scene_question_answering', ['llama3_70b_judge'])
446
- else:
447
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['llama3_70b_judge'])
448
- draw('asu', 'audio_scene_question_answering', filter_1, 'llama3_70b_judge')
449
-
450
-
451
- def emotion_recognition():
452
- st.title("Task: Emotion Recognition")
453
-
454
- sum = ['Overall']
455
-
456
- filters_levelone = sum + er_datasets
457
-
458
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
459
-
460
- with left:
461
- filter_1 = st.selectbox('Dataset', filters_levelone)
462
-
463
- if filter_1:
464
- if filter_1 in sum:
465
- sum_table_mulit_metrix('emotion_recognition', ['llama3_70b_judge'])
466
- else:
467
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['llama3_70b_judge'])
468
- draw('vu', 'emotion_recognition', filter_1, 'llama3_70b_judge')
469
-
470
-
471
- def accent_recognition():
472
- st.title("Task: Accent Recognition")
473
-
474
- sum = ['Overall']
475
-
476
- filters_levelone = sum + ar_datasets
477
-
478
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
479
-
480
- with left:
481
- filter_1 = st.selectbox('Dataset', filters_levelone)
482
-
483
-
484
- if filter_1:
485
- if filter_1 in sum:
486
- sum_table_mulit_metrix('accent_recognition', ['llama3_70b_judge'])
487
- else:
488
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['llama3_70b_judge'])
489
- draw('vu', 'accent_recognition', filter_1, 'llama3_70b_judge')
490
-
491
-
492
- def gender_recognition():
493
- st.title("Task: Gender Recognition")
494
-
495
- sum = ['Overall']
496
-
497
- filters_levelone = sum + gr_datasets
498
-
499
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
500
-
501
- with left:
502
- filter_1 = st.selectbox('Dataset', filters_levelone)
503
-
504
- if filter_1:
505
- if filter_1 in sum:
506
- sum_table_mulit_metrix('gender_recognition', ['llama3_70b_judge'])
507
- else:
508
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['llama3_70b_judge'])
509
- draw('vu', 'gender_recognition', filter_1, 'llama3_70b_judge')
510
-
511
-
512
- def music_understanding():
513
- st.title("Task: Music Understanding - MCQ Questions")
514
-
515
- sum = ['Overall']
516
-
517
- filters_levelone = sum + music_datasets
518
-
519
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
520
-
521
- with left:
522
- filter_1 = st.selectbox('Dataset', filters_levelone)
523
-
524
- if filter_1:
525
- if filter_1 in sum:
526
- sum_table_mulit_metrix('music_understanding', ['llama3_70b_judge'])
527
- else:
528
- dataset_contents(dataset_diaplay_information[filter_1], metrics_info['llama3_70b_judge'])
529
- draw('vu', 'music_understanding', filter_1, 'llama3_70b_judge')
530
-
531
-
532
- def under_development():
533
- st.title("Task: Under Development")
534
-
535
- filters_levelone = non_wer_development_datasets + wer_development_datasets
536
-
537
- left, center, _, middle, right = st.columns([0.4, 0.2, 0.2, 0.2 ,0.2])
538
-
539
- with left:
540
- filter_1 = st.selectbox('Dataset', filters_levelone)
541
-
542
- dataset_contents(dataset_diaplay_information[filter_1], 'under_development')
543
-
544
- # = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = =
545
-
546
- '''
547
- Show Dataset Examples
548
- '''
549
-
550
- # Initialize a session state variable for toggling the chart visibility
551
- if "show_dataset_examples" not in st.session_state:
552
- st.session_state.show_dataset_examples = False
553
-
554
- # Create a button to toggle visibility
555
- if st.button("Show Dataset Examples"):
556
- st.session_state.show_dataset_examples = not st.session_state.show_dataset_examples
557
-
558
- if st.session_state.show_dataset_examples:
559
-
560
- # st.markdown('To be implemented')
561
-
562
- # # if dataset_name in ['Earnings21-Test', 'Earnings22-Test', 'Tedlium3-Test', 'Tedlium3-Long-form-Test']:
563
- if filter_1 in []:
564
- pass
565
- else:
566
- try:
567
- show_dataset_examples(filter_1)
568
- except:
569
- st.markdown('To be implemented')
570
- # = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = =
571
-
572
- if filter_1 in wer_development_datasets:
573
- draw('vu', 'under_development_wer', filter_1, 'wer')
574
-
575
- elif filter_1 in non_wer_development_datasets:
576
- draw('vu', 'under_development_llama3_70b_judge', filter_1, 'llama3_70b_judge')
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
app/show_examples.py DELETED
@@ -1,199 +0,0 @@
1
- import streamlit as st
2
- import datasets
3
- import numpy as np
4
- import io
5
- import soundfile as sf
6
-
7
- import html
8
-
9
- from app.content import displayname2datasetname
10
-
11
- def show_dataset_examples(display_name):
12
- st.divider()
13
- dataset_name = displayname2datasetname[display_name]
14
- sample_folder = f"./examples/{dataset_name}"
15
-
16
- # load dataset
17
- dataset = datasets.load_from_disk(sample_folder)
18
-
19
- for index in range(len(dataset)):
20
- with st.container():
21
- st.markdown(f'##### Example-{index+1}')
22
- col1, col2 = st.columns([0.3, 0.7], vertical_alignment="center")
23
-
24
- with col1:
25
- # Convert the NumPy array to a WAV file in memory
26
- bytes_io = io.BytesIO()
27
- sf.write(bytes_io, dataset[index]['context']['audio']['array'], dataset[index]['context']['audio']['sampling_rate'], format='WAV')
28
- bytes_io.seek(0)
29
- # Play audio in Streamlit
30
- st.audio(bytes_io, format='audio/wav')
31
- # st.audio(f'{sample_folder}/sample_{index}.wav', format="audio/wav")
32
-
33
- if dataset_name in ['CN-College-Listen-MCQ-Test', 'DREAM-TTS-MCQ-Test']:
34
-
35
- choices = dataset[index]['other_attributes']['choices']
36
- if isinstance(choices, str):
37
- choices_text = choices
38
- elif isinstance(choices, list):
39
- choices_text = ' '.join(i for i in choices)
40
-
41
- question_text = f"""{dataset[index]['instruction']['text']} {choices_text}"""
42
- else:
43
- question_text = f"""{dataset[index]['instruction']['text']}"""
44
-
45
- question_text = html.escape(question_text)
46
-
47
- # with st.container():
48
- with col2:
49
- custom_css = """
50
- <style>
51
- .my-container-table, p.my-container-text {
52
- background-color: #fcf8dc;
53
- padding: 10px;
54
- border-radius: 5px;
55
- font-size: 13px;
56
- # height: 50px;
57
- word-wrap: break-word
58
- }
59
- </style>
60
- """
61
- st.markdown(custom_css, unsafe_allow_html=True)
62
-
63
- body_details = f"""<table style="table-layout: fixed; width:100%">
64
- <thead>
65
- <tr style="text-align: center;">
66
- <th style="width:50%">PROMPT</th>
67
- <th style="width:50%">ANSWER</th>
68
- </tr>
69
- <tr>
70
- <td><b>{html.escape(question_text.replace('(A)', '<br>(A)').replace('(B)', '<br>(B)').replace('(C)', '<br>(C)'))}
71
- </td>
72
- <td><b>{html.escape(dataset[index]['answer']['text'])}
73
- </td>
74
- </tr>
75
- </thead>
76
- </table>"""
77
-
78
- st.markdown(f"""<div class="my-container-table">
79
- {body_details}
80
- </div>""", unsafe_allow_html=True)
81
-
82
- st.text("")
83
-
84
- st.divider()
85
-
86
-
87
- def show_examples(category_name, dataset_name, model_lists, display_model_names):
88
- st.divider()
89
- sample_folder = f"./examples/{category_name}/{dataset_name}"
90
-
91
- dataset = datasets.load_from_disk(sample_folder)
92
-
93
- for index in range(len(dataset)):
94
- with st.container():
95
- st.markdown(f'##### Example-{index+1}')
96
- col1, col2 = st.columns([0.3, 0.7], vertical_alignment="center")
97
-
98
- # with col1:
99
- st.audio(f'{sample_folder}/sample_{index}.wav', format="audio/wav")
100
-
101
- if dataset_name in ['CN-College-Listen-MCQ-Test', 'DREAM-TTS-MCQ-Test']:
102
-
103
- choices = dataset[index]['other_attributes']['choices']
104
- if isinstance(choices, str):
105
- choices_text = choices
106
- elif isinstance(choices, list):
107
- choices_text = ' '.join(i for i in choices)
108
-
109
- question_text = f"""{dataset[index]['instruction']['text']} {choices_text}"""
110
- else:
111
- question_text = f"""{dataset[index]['instruction']['text']}"""
112
-
113
- question_text = html.escape(question_text)
114
-
115
- # st.divider()
116
- with st.container():
117
- custom_css = """
118
- <style>
119
- .my-container-table, p.my-container-text {
120
- background-color: #fcf8dc;
121
- padding: 10px;
122
- border-radius: 5px;
123
- font-size: 13px;
124
- # height: 50px;
125
- word-wrap: break-word
126
- }
127
- </style>
128
- """
129
- st.markdown(custom_css, unsafe_allow_html=True)
130
-
131
- model_lists.sort()
132
-
133
- s = f"""<tr>
134
- <td><b>REFERENCE</td>
135
- <td><b>{html.escape(question_text.replace('(A)', '<br>(A)').replace('(B)', '<br>(B)').replace('(C)', '<br>(C)'))}
136
- </td>
137
- <td><b>{html.escape(dataset[index]['answer']['text'])}
138
- </td>
139
- </tr>
140
- """
141
- if dataset_name in ['CN-College-Listen-MCQ-Test', 'DREAM-TTS-MCQ-Test']:
142
- for model in model_lists:
143
- try:
144
-
145
- model_prediction = dataset[index][model]['model_prediction']
146
- model_prediction = model_prediction.replace('<','').replace('>','').replace('\n','(newline)').replace('*','')
147
-
148
- s += f"""<tr>
149
- <td>{display_model_names[model]}</td>
150
- <td>
151
- {dataset[index][model]['text'].replace('Choices:', '<br>Choices:').replace('(A)', '<br>(A)').replace('(B)', '<br>(B)').replace('(C)', '<br>(C)')
152
- }
153
- </td>
154
- <td>{html.escape(model_prediction)}</td>
155
- </tr>"""
156
- except:
157
- print(f"{model} is not in {dataset_name}")
158
- continue
159
- else:
160
- for model in model_lists:
161
-
162
- print(dataset[index][model]['model_prediction'])
163
-
164
- try:
165
-
166
- model_prediction = dataset[index][model]['model_prediction']
167
- model_prediction = model_prediction.replace('<','').replace('>','').replace('\n','(newline)').replace('*','')
168
-
169
- s += f"""<tr>
170
- <td>{display_model_names[model]}</td>
171
- <td>{html.escape(dataset[index][model]['text'])}</td>
172
- <td>{html.escape(model_prediction)}</td>
173
- </tr>"""
174
- except:
175
- print(f"{model} is not in {dataset_name}")
176
- continue
177
-
178
-
179
- body_details = f"""<table style="table-layout: fixed; width:100%">
180
- <thead>
181
- <tr style="text-align: center;">
182
- <th style="width:20%">MODEL</th>
183
- <th style="width:30%">QUESTION</th>
184
- <th style="width:50%">MODEL PREDICTION</th>
185
- </tr>
186
- {s}
187
- </thead>
188
- </table>"""
189
-
190
- st.markdown(f"""<div class="my-container-table">
191
- {body_details}
192
- </div>""", unsafe_allow_html=True)
193
-
194
- st.text("")
195
-
196
- st.divider()
197
-
198
-
199
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
app/summarization.py DELETED
@@ -1,133 +0,0 @@
1
- import streamlit as st
2
- import pandas as pd
3
- import numpy as np
4
- from streamlit_echarts import st_echarts
5
- from streamlit.components.v1 import html
6
- # from PIL import Image
7
- from app.show_examples import *
8
- from app.content import *
9
-
10
- import pandas as pd
11
- from typing import List
12
-
13
- from model_information import get_dataframe
14
-
15
- info_df = get_dataframe()
16
-
17
-
18
- def sum_table_mulit_metrix(task_name, metrics_lists: List[str]):
19
-
20
- # combine chart data from multiple sources
21
- chart_data = pd.DataFrame()
22
- for metrics in metrics_lists:
23
- folder = f"./results_organized/{metrics}"
24
- data_path = f'{folder}/{task_name.lower()}.csv'
25
- one_chart_data = pd.read_csv(data_path).round(3)
26
- if len(chart_data) == 0:
27
- chart_data = one_chart_data
28
- else:
29
- chart_data = pd.merge(chart_data, one_chart_data, on='Model', how='outer')
30
-
31
- selected_columns = [i for i in chart_data.columns if i != 'Model']
32
- # TODO: temp code. delete this after ytb tamil vs is fully tested.
33
- _columns_to_exclude_from_average = ["ytb_asr_batch3_tamil_v2"]
34
- selected_columns = [col for col in selected_columns if col not in _columns_to_exclude_from_average]
35
- chart_data['Average'] = chart_data[selected_columns].mean(axis=1)
36
-
37
- # Update dataset name in table
38
- chart_data = chart_data.rename(columns=datasetname2diaplayname)
39
-
40
- st.markdown("""
41
- <style>
42
- .stMultiSelect [data-baseweb=select] span {
43
- max-width: 800px;
44
- font-size: 0.9rem;
45
- background-color: #3C6478 !important; /* Background color for selected items */
46
- color: white; /* Change text color */
47
- back
48
- }
49
- </style>
50
- """, unsafe_allow_html=True)
51
-
52
- # remap model names
53
- display_model_names = {key.strip() :val.strip() for key, val in zip(info_df['Original Name'], info_df['Proper Display Name'])}
54
- chart_data['model_show'] = chart_data['Model'].map(lambda x: display_model_names.get(x, x))
55
-
56
- models = st.multiselect("Please choose the model",
57
- sorted(chart_data['model_show'].tolist()),
58
- default = sorted(chart_data['model_show'].tolist()),
59
- )
60
- # TODO: delete this after ytb tamil v2 is fully utilized.
61
- if task_name == 'asr_tamil':
62
- chart_data = chart_data[chart_data['model_show'].isin(models)]
63
- else:
64
- chart_data = chart_data[chart_data['model_show'].isin(models)].dropna(axis=0)
65
-
66
- if len(chart_data) == 0: return
67
-
68
- # = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = = =
69
- '''
70
- Show Table
71
- '''
72
- with st.container():
73
- st.markdown(f'##### TABLE')
74
-
75
- model_link = {key.strip(): val for key, val in zip(info_df['Proper Display Name'], info_df['Link'])}
76
-
77
- chart_data['model_link'] = chart_data['model_show'].map(model_link)
78
-
79
- tabel_columns = [i for i in chart_data.columns if i not in ['Model', 'model_show']]
80
- column_to_front = 'Average'
81
- new_order = [column_to_front] + [col for col in tabel_columns if col != column_to_front]
82
-
83
- chart_data_table = chart_data[['model_show'] + new_order]
84
-
85
-
86
- # Format numeric columns to 2 decimal places
87
- target_column = chart_data_table.columns[1]
88
- chart_data_table.loc[:, target_column] = chart_data_table[target_column].apply(lambda x: round(float(x), 3) if isinstance(float(x), (int, float)) else float(x))
89
-
90
- if metrics in ['wer']:
91
- ascend = True
92
- else:
93
- ascend= False
94
-
95
- chart_data_table = chart_data_table.sort_values(
96
- by=['Average'],
97
- ascending=ascend
98
- ).reset_index(drop=True)
99
-
100
- # Highlight the best performing model
101
- def highlight_first_element(x):
102
- # Create a DataFrame with the same shape as the input
103
- df_style = pd.DataFrame('', index=x.index, columns=x.columns)
104
- # Apply background color to the first element in row 0 (df[0][0])
105
- # df_style.iloc[0, 1] = 'background-color: #b0c1d7; color: white'
106
- df_style.iloc[0, 1] = 'background-color: #b0c1d7'
107
-
108
- return df_style
109
-
110
-
111
- styled_df = chart_data_table.style.format(
112
- {
113
- chart_data_table.columns[i]: "{:.3f}" for i in range(1, len(chart_data_table.columns) - 1)
114
- }
115
- ).apply(
116
- highlight_first_element, axis=None
117
- )
118
-
119
- st.dataframe(
120
- styled_df,
121
- column_config={
122
- 'model_show': 'Model',
123
- chart_data_table.columns[1]: {'alignment': 'left'},
124
- "model_link": st.column_config.LinkColumn(
125
- "Model Link",
126
- ),
127
- },
128
- hide_index=True,
129
- use_container_width=True
130
- )
131
-
132
- # Only report the last metrics
133
- st.markdown(f'###### Metric: {metrics_info[metrics]}')
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
assets/index-C9x3RgO7.js ADDED
The diff for this file is too large to render. See raw diff
 
assets/index-D38mSSRC.css ADDED
@@ -0,0 +1 @@
 
 
1
+ *,*:before,*:after{box-sizing:border-box;margin:0;padding:0}:root{--bg:#f4f6fa;--surface:#fff;--border:#e2e6ef;--text:#1a2332;--text2:#5a6377;--text3:#8892a4;--accent:#3C6478;--accent-light:#eaf2f7;--best:#dbeafe;--radius:10px;--shadow:0 1px 3px rgba(0,0,0,.06);--sidebar-w:260px;--font:-apple-system,BlinkMacSystemFont,"Segoe UI",Roboto,"Helvetica Neue",Arial,sans-serif}html{font-size:14px}body{font-family:var(--font);color:var(--text);background:var(--bg);line-height:1.55}a{color:var(--accent);text-decoration:none}a:hover{text-decoration:underline}.layout{display:flex;min-height:100vh}.sidebar{width:var(--sidebar-w);background:linear-gradient(180deg,#1a2332,#243447);color:#c8d1dc;position:fixed;top:0;left:0;bottom:0;overflow-y:auto;display:flex;flex-direction:column;z-index:10}.sidebar h2{padding:20px 18px 8px;font-size:1.1rem;color:#fff;letter-spacing:.02em}.sidebar-links{margin-top:12px}.main{margin-left:var(--sidebar-w);flex:1;padding:28px 36px 60px;max-width:1280px;min-width:0;overflow-x:hidden}.nav-group-label{padding:10px 18px 6px;font-size:.75rem;text-transform:uppercase;letter-spacing:.07em;color:#8a9bb0;font-weight:700;cursor:pointer;display:flex;align-items:center;gap:6px;-webkit-user-select:none;user-select:none;border-left:3px solid transparent;transition:background .12s,color .12s,border-color .12s}.nav-group-label:hover{color:#b4c5d8;background:#ffffff0a}.nav-group-label .arrow{font-size:.6rem;transition:transform .15s}.nav-group-label .arrow.open{transform:rotate(90deg)}.nav-group-label.expandable{padding-bottom:4px}.nav-group-label.clickable{padding:10px 18px}.nav-group-label.clickable:hover{background:#ffffff0f;color:#c8d6e5}.nav-group-label.clickable.active{background:#ffffff1a;color:#fff;border-left-color:var(--accent)}.nav-item.nav-dashboard{padding:7px 18px;font-size:.85rem;font-weight:600}.nav-item{display:block;padding:6px 18px 6px 32px;font-size:.82rem;color:#7e93aa;cursor:pointer;transition:background .12s,color .12s;border-left:3px solid transparent}.nav-item:hover{background:#ffffff0d;color:#b4c5d8}.nav-item.active{background:#ffffff14;color:#e8edf3;border-left-color:var(--accent);font-weight:600}.hero{background:linear-gradient(135deg,#1a2332,#2d4a5e 50%,#3c6478);border-radius:var(--radius);padding:36px 40px;color:#fff;margin-bottom:24px}.hero h1{font-size:2rem;font-weight:800;margin-bottom:6px}.hero .links{display:flex;gap:12px;flex-wrap:wrap}.hero .links a{color:#a0cfec;font-size:.88rem}.stat-row{display:flex;gap:14px;flex-wrap:wrap;margin-bottom:20px}.stat-card{background:var(--surface);border:1px solid var(--border);border-radius:var(--radius);padding:14px 20px;min-width:120px;flex:1;box-shadow:var(--shadow)}.stat-card .label{font-size:.72rem;text-transform:uppercase;letter-spacing:.04em;color:var(--text2);font-weight:700;margin-bottom:2px}.stat-card .value{font-size:1.5rem;font-weight:700;color:var(--text)}.card{background:var(--surface);border:1px solid var(--border);border-radius:var(--radius);padding:20px 24px;margin-bottom:16px;box-shadow:var(--shadow);overflow:hidden}.section-title{font-size:1rem;font-weight:700;margin-bottom:12px;color:var(--text)}.collapsible-header{cursor:pointer;display:flex;align-items:center;gap:8px;font-weight:700;font-size:.92rem;color:var(--text);-webkit-user-select:none;user-select:none;padding:12px 0}.collapsible-header:hover{color:var(--accent)}.collapsible-body{padding-bottom:14px}.table-wrap{overflow-x:auto;margin-bottom:16px}table.lb{width:100%;border-collapse:separate;border-spacing:0;font-size:.85rem}table.lb th{position:sticky;top:0;background:var(--bg);padding:10px 12px;text-align:left;font-weight:700;color:var(--text2);white-space:nowrap;border-bottom:2px solid var(--border);font-size:.78rem;text-transform:uppercase;letter-spacing:.03em}table.lb td{padding:9px 12px;border-bottom:1px solid var(--border);white-space:nowrap}table.lb tr:hover td{background:var(--accent-light)}table.lb tr.best td{background:var(--best)}table.lb td.num{text-align:right;font-variant-numeric:tabular-nums}table.lb .model-name{font-weight:600;max-width:320px;overflow:hidden;text-overflow:ellipsis}table.lb .rank{color:var(--text3);font-weight:600;width:36px;text-align:center}table.lb th.sortable{cursor:pointer;-webkit-user-select:none;user-select:none}table.lb th.sortable:hover{color:var(--accent)}table.lb th.has-stats{position:relative}.col-header-content{display:inline-flex;align-items:center;gap:2px}.sort-indicator{font-size:.7rem;color:var(--text3);opacity:.4;transition:opacity .15s}.sort-indicator.active{opacity:1;color:var(--accent)}.col-tooltip{display:none;position:absolute;left:0;top:100%;z-index:20;background:var(--surface);border:1px solid var(--border);border-radius:8px;padding:14px 16px;min-width:260px;box-shadow:0 4px 16px #0000001f;font-weight:400;text-transform:none;letter-spacing:0;font-size:.82rem;color:var(--text);line-height:1.6;white-space:normal}table.lb th.has-stats:hover .col-tooltip{display:block}.col-tooltip .tt-label{font-size:.68rem;text-transform:uppercase;letter-spacing:.05em;color:var(--text3);font-weight:700;margin-bottom:2px}.col-tooltip .tt-val{font-size:1rem;font-weight:700;color:var(--text);margin-bottom:8px}.col-tooltip .tt-desc{font-size:.8rem;color:var(--text2);margin-bottom:8px}.col-tooltip .tt-stats{display:grid;grid-template-columns:1fr 1fr;gap:4px 14px;font-size:.78rem}.col-tooltip .tt-stat-label{color:var(--text3)}.col-tooltip .tt-stat-val{font-weight:600}.col-tooltip .tt-link{margin-top:8px;font-size:.78rem}.dataset-detail-item{border-bottom:1px solid var(--border);padding:12px 0}.dataset-detail-item:last-child{border-bottom:none}.dataset-detail-header{display:flex;align-items:baseline;gap:4px;flex-wrap:wrap}.dataset-stats-row{display:flex;align-items:flex-start;gap:20px;margin-top:6px;flex-wrap:wrap}.dataset-stats-numbers{display:flex;gap:16px;flex-wrap:wrap;font-size:.78rem;color:var(--text3)}.mini-hist{flex-shrink:0}.mini-hist-bars{display:flex;align-items:flex-end;gap:3px;height:48px}.mini-hist-col{display:flex;flex-direction:column;align-items:center;width:40px}.mini-hist-bar{width:100%;background:var(--accent);border-radius:2px 2px 0 0;min-height:1px;transition:height .2s}.mini-hist-label{font-size:.58rem;color:var(--text3);margin-top:2px;white-space:nowrap}.examples-section{margin-top:8px;background:var(--bg);border-radius:8px;padding:10px 14px}.examples-header{display:flex;align-items:center;gap:10px;margin-bottom:8px}.examples-title{font-size:.78rem;font-weight:700;color:var(--text2);text-transform:uppercase;letter-spacing:.04em}.examples-nav{display:flex;gap:4px}.examples-nav-btn{width:22px;height:22px;border-radius:50%;border:1px solid var(--border);background:var(--surface);font-size:.72rem;font-weight:700;color:var(--text2);cursor:pointer;display:flex;align-items:center;justify-content:center;transition:all .12s}.examples-nav-btn:hover{border-color:var(--accent);color:var(--accent)}.examples-nav-btn.active{background:var(--accent);color:#fff;border-color:var(--accent)}.examples-body{display:flex;flex-direction:column;gap:6px}.example-row{display:flex;align-items:flex-start;gap:10px;font-size:.82rem}.example-label{font-weight:700;color:var(--text3);font-size:.72rem;text-transform:uppercase;letter-spacing:.04em;min-width:64px;padding-top:2px;flex-shrink:0}.example-text{color:var(--text2);line-height:1.5;word-break:break-word;overflow-wrap:anywhere;min-width:0;flex:1}.examples-body audio{height:32px;max-width:300px}.metric-badge{display:inline-block;background:var(--accent-light);color:var(--accent);font-size:.74rem;font-weight:700;padding:3px 10px;border-radius:20px;margin-left:8px}.mobile-menu-btn{display:none;position:fixed;top:12px;left:12px;z-index:30;width:40px;height:40px;border-radius:8px;border:none;background:var(--accent);color:#fff;font-size:1.3rem;cursor:pointer;align-items:center;justify-content:center;box-shadow:0 2px 8px #0000002e}.sidebar-overlay{display:none;position:fixed;inset:0;background:#00000073;z-index:9}@media(max-width:860px){.mobile-menu-btn{display:flex}.sidebar{width:var(--sidebar-w);transform:translate(-100%);transition:transform .25s ease}.sidebar.open{transform:translate(0)}.sidebar-overlay.open{display:block}.main{margin-left:0;padding:52px 16px 32px}}
examples/cna_test/example_0.wav ADDED
Binary file (65 kB). View file
 
examples/{ytb_asr_batch3_tamil/data-00000-of-00001.arrow → cna_test/example_1.wav} RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:610511896b4e2b7c2ade028585ae16c3d8e4f9cdbf7767e2fd1d23005f307a89
3
- size 890760
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d671a9ef97a17dcbc93dde2bb91ac873d8f2412b6b0b76cc63b2f241a264f1ab
3
+ size 117868
examples/{ytb_sds_batch3_chinese/data-00000-of-00001.arrow → commonvoice_17_id_asr/example_0.wav} RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6118f3210caa8872b421495bb034b388a63cf8b14b6dc1e2e1f72347fea4e1f6
3
- size 260440
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2873873174ef2b47c9768834427313dc06c4787d5e2ece5559592074d1b2cac5
3
+ size 262700
examples/{ytb_asr_batch3_chinese/data-00000-of-00001.arrow → commonvoice_17_id_asr/example_1.wav} RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fca176caa14fa2d13d2fdf96c99a0e8dfec17e2b83af14b78a1d18a0d311ed1c
3
- size 152264
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:49ee064f2f06e63ca0c0f328461051f8da550af602b4ba684bdc2d8fd562c1f7
3
+ size 124460
examples/{ytb_asr_batch3_malay/data-00000-of-00001.arrow → commonvoice_17_ta_asr/example_0.wav} RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9659b947c6b10c86c127690ac15403d40f9e71e541a4bb4391545d6329c7313f
3
- size 702736
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fb926eb3c960beee6ff08bca9498950bf1c34376419962c5a9029003942db15d
3
+ size 146732
examples/commonvoice_17_ta_asr/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:daa9812c5ae8ff9c4645af3393c241a7b3b8609d210e80141da797591c0ec35e
3
+ size 163628
examples/commonvoice_17_th_asr/example_0.wav ADDED
Binary file (98 kB). View file
 
examples/commonvoice_17_th_asr/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b84d3b0d0330d233bab30e235d4864fcd91a202cb1b124d1bf04340a75d3c40b
3
+ size 109484
examples/commonvoice_17_vi_asr/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3232ae3fd1e9d8eb79cbf9ede448070b2813ea53d3a425101f9c653219f29d86
3
+ size 104876
examples/commonvoice_17_vi_asr/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51c5e36f8ce7c09aeedf461accbd39fdda7bef172d24abf4c54a5d59031692ac
3
+ size 145196
examples/commonvoice_zh_asr/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:99da77bb3103c130a65cfbf6806bd368ffa8db63793c543c1743932335220b00
3
+ size 231980
examples/commonvoice_zh_asr/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e8ca40944a0d287bdc966f9f54df3104f493ade7b8bf7105217429a86c59c1b7
3
+ size 176300
examples/fleurs_tamil_ta_30_asr/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6457a73e386ab3654f79548791b6ff73292e5a315654e266ea886ab068cc0781
3
+ size 480044
examples/fleurs_tamil_ta_30_asr/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d9561b9a63d66cc4fe3271a0ea7028ff005027080081ce1eabc554209ff180e6
3
+ size 441644
examples/gigaspeech2_id_test/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bfec7fba413fc8e7da61d5c24d73bdd77e016b297436ac51bb8a624c949bd8ac
3
+ size 220266
examples/gigaspeech2_id_test/example_1.wav ADDED
Binary file (87.8 kB). View file
 
examples/gigaspeech2_th_test/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0982da6bb864511312c1589a125866e6fb44f06197c8b42771caceb99e6118ae
3
+ size 295722
examples/gigaspeech2_th_test/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b347da5f01c383d5ab33d530a08824bb1dfa3a5ca9fddc31e0091825a6f079ce
3
+ size 150540
examples/gigaspeech2_vi_test/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:58f881efda72405f79eb391f17866d34004a124e96fc85653c928fd8d8a5370c
3
+ size 355812
examples/gigaspeech2_vi_test/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0c0a164361ecb493b76cbb26d5747b4dfc2a4d6623067b677d6ad108991bc6f0
3
+ size 250382
examples/idpc_short_test/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0c8d813d57338bb5db29d1b6ec2f428c94265da032f4feba06df05690c0b5ac2
3
+ size 846124
examples/idpc_short_test/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ad8dad30f3dddeaad7e32336841a2cb2a15d3761c38dd62ad2ec6f49d956dd68
3
+ size 418284
examples/imda_ar_dialogue/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a25e546945a11abecef9ea36f6b4ab34d4949803cad06faf3f45857536dceba2
3
+ size 897644
examples/imda_ar_dialogue/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:91a6b03725839579b6806bd9a28057a51e8a9cd28f359fe09cfa321e6f41d4ef
3
+ size 867124
examples/imda_ar_sentence/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7e60ddc2b12238cbcad5dcc611754ab995ab8cd5789ef3ab13a095c9340e5c17
3
+ size 158124
examples/imda_ar_sentence/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f3235a16d5654c470db58d787b54d4b7b79275d4c6566007ec80ee6693ba1a2a
3
+ size 162444
examples/imda_part1_asr_test/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:137d9c12879d7432c8665ad938e456027d621a40c3b13d4dd6f715a50177602d
3
+ size 127724
examples/imda_part1_asr_test/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:150e574c811a6bed67870e260250383a9821db884c2931a1cdee2c4c8e165f0f
3
+ size 176812
examples/imda_part2_asr_test/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cb4bf2857925ed079a40a542bc9382154a02ccf08c4e4e0a037eeaf67ed62c3f
3
+ size 131468
examples/imda_part2_asr_test/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a23f686148140282b7b39002f81ec9df7142d9492bb0c8cc7da84c0b51cccfbb
3
+ size 179596
examples/imda_part3_30s_asr_test/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:125993396252ba54433006e3cf3e248b4c15e87e4a2d041f5853d9ddbd8d7cc7
3
+ size 929644
examples/imda_part3_30s_asr_test/example_1.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9834aca6dd6a7783589f7f437aeff2848d0cd4125b236fd3735eb674f880a610
3
+ size 826828
examples/imda_part3_30s_ds_human_test/example_0.wav ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:57c2c8c700f4469ad1673408ff10c7a23eac1e206bfc6daf45597ef5176daeea
3
+ size 955596