{ "tasks": [ { "key": "ASR-English", "title": "Task: Automatic Speech Recognition - English", "taskName": "asr_english", "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. WavLLM's error rate on asr_cv21_en_30 (0.489) reflects output-language drift rather than transcription accuracy: on roughly 23% of Common Voice clips it answers in German or Chinese despite prompts specifying English, and those answers are usually correct translations of the audio. Its rate on the other English sets in this table is 0.05-0.11. The drift tracks the audio, not the prompt -- it is 21-36% across all ten prompt variants on that dataset and under 2% on the others.", "ascending": true, "datasets": [ { "display": "LibriSpeech-Clean", "internal": "librispeech_test_clean", "description": "A clean, high-quality testset of the LibriSpeech dataset, used for ASR testing.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/librispeech_test_clean_v2", "stats": { "num_rows": 2617, "audio_length": { "min": 1.28, "max": 34.95, "mean": 7.43, "median": 5.79, "std": 5.15, "total_hours": 5.4, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 2608, 9, 0, 0, 0 ] } } } }, { "display": "LibriSpeech-Other", "internal": "librispeech_test_other", "description": "A more challenging, noisier testset of the LibriSpeech dataset for ASR testing.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/librispeech_test_other_v2", "stats": { "num_rows": 2935, "audio_length": { "min": 1.47, "max": 34.51, "mean": 6.55, "median": 5.16, "std": 4.43, "total_hours": 5.34, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 2928, 7, 0, 0, 0 ] } } } }, { "display": "CommonVoice-15-EN", "internal": "common_voice_15_en_test", "description": "Test set from the Common Voice project, which is a crowd-sourced, multilingual speech dataset.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/common_voice_15_en_test_v2", "stats": { "num_rows": 16348, "audio_length": { "min": 1.34, "max": 105.67, "mean": 5.93, "median": 5.74, "std": 2.35, "total_hours": 26.95, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 16343, 5, 0, 0, 0 ] } } } }, { "display": "Peoples-Speech", "internal": "peoples_speech_test", "description": "A large-scale, open-source speech recognition dataset, with diverse accents and domains.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/peoples_speech_test_v2", "stats": { "num_rows": 32603, "audio_length": { "min": 1.0, "max": 99.91, "mean": 6.54, "median": 5.0, "std": 5.33, "total_hours": 59.2, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 32449, 154, 0, 0, 0 ] } } } }, { "display": "GigaSpeech-1", "internal": "gigaspeech_test", "description": "A large-scale ASR dataset with diverse audio sources like podcasts, interviews, etc.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/gigaspeech_test_v2", "stats": { "num_rows": 18650, "audio_length": { "min": 1.0, "max": 22.02, "mean": 6.77, "median": 6.54, "std": 3.53, "total_hours": 35.1, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 18650, 0, 0, 0, 0 ] } } } }, { "display": "Earnings-21", "internal": "earnings21_test", "description": "ASR test dataset focused on earnings calls from 2021, with professional speech and financial jargon.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/earnings21_test", "stats": { "num_rows": 44, "audio_length": { "min": 1097.1, "max": 5740.61, "mean": 3212.41, "median": 3280.62, "std": 1141.07, "total_hours": 39.26, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 0, 0, 0, 6, 38 ] } } } }, { "display": "Earnings-22", "internal": "earnings22_test", "description": "Similar to Earnings21, but covering earnings calls from 2022.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/earnings22_test", "stats": { "num_rows": 125, "audio_length": { "min": 874.73, "max": 7407.05, "mean": 3452.71, "median": 3552.12, "std": 1252.95, "total_hours": 119.89, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 0, 0, 0, 16, 109 ] } } } }, { "display": "TED-LIUM-3", "internal": "tedlium3_test", "description": "A test set derived from TED talks, covering diverse speakers and topics.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/tedlium3_test_v2", "stats": { "num_rows": 1142, "audio_length": { "min": 1.07, "max": 32.55, "mean": 8.24, "median": 8.25, "std": 4.27, "total_hours": 2.61, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 1141, 1, 0, 0, 0 ] } } } }, { "display": "TED-LIUM-3-LongForm", "internal": "tedlium3_long_form_test", "description": "A longer version of the TED-LIUM dataset, containing extended audio samples. This poses challenges to existing fusion methods in handling long audios. However, it provides benchmark for future development.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/tedlium3_long_form_test_v2", "stats": { "num_rows": 11, "audio_length": { "min": 340.96, "max": 1596.73, "mean": 856.51, "median": 921.06, "std": 407.16, "total_hours": 2.62, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 0, 0, 0, 11, 0 ] } } } }, { "display": "Bloomspeech-EN-30 [SEA]", "internal": "asr_bloomspeech_en_30", "description": "BloomSpeech English ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Cv21-EN-30 [SEA]", "internal": "asr_cv21_en_30", "description": "CV21 English ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "ESD-EN-30 [SEA]", "internal": "asr_esd_en_30", "description": "ESD English ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} } ], "data": { "columns": [ { "key": "librispeech_test_clean", "display": "LibriSpeech-Clean", "description": "A clean, high-quality testset of the LibriSpeech dataset, used for ASR testing.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/librispeech_test_clean_v2", "isWer": true }, { "key": "librispeech_test_other", "display": "LibriSpeech-Other", "description": "A more challenging, noisier testset of the LibriSpeech dataset for ASR testing.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/librispeech_test_other_v2", "isWer": true }, { "key": "common_voice_15_en_test", "display": "CommonVoice-15-EN", "description": "Test set from the Common Voice project, which is a crowd-sourced, multilingual speech dataset.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/common_voice_15_en_test_v2", "isWer": true }, { "key": "peoples_speech_test", "display": "Peoples-Speech", "description": "A large-scale, open-source speech recognition dataset, with diverse accents and domains.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/peoples_speech_test_v2", "isWer": true }, { "key": "gigaspeech_test", "display": "GigaSpeech-1", "description": "A large-scale ASR dataset with diverse audio sources like podcasts, interviews, etc.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/gigaspeech_test_v2", "isWer": true }, { "key": "earnings21_test", "display": "Earnings-21", "description": "ASR test dataset focused on earnings calls from 2021, with professional speech and financial jargon.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/earnings21_test", "isWer": true }, { "key": "earnings22_test", "display": "Earnings-22", "description": "Similar to Earnings21, but covering earnings calls from 2022.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/earnings22_test", "isWer": true }, { "key": "tedlium3_test", "display": "TED-LIUM-3", "description": "A test set derived from TED talks, covering diverse speakers and topics.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/tedlium3_test_v2", "isWer": true }, { "key": "tedlium3_long_form_test", "display": "TED-LIUM-3-LongForm", "description": "A longer version of the TED-LIUM dataset, containing extended audio samples. This poses challenges to existing fusion methods in handling long audios. However, it provides benchmark for future development.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/tedlium3_long_form_test_v2", "isWer": true }, { "key": "asr_bloomspeech_en_30", "display": "Bloomspeech-EN-30 [SEA]", "description": "BloomSpeech English ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_cv21_en_30", "display": "Cv21-EN-30 [SEA]", "description": "CV21 English ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_esd_en_30", "display": "ESD-EN-30 [SEA]", "description": "ESD English ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true } ], "rows": [ { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_bloomspeech_en_30": 0.05087901, "asr_cv21_en_30": 0.06627517, "asr_esd_en_30": 0.02791435, "average": 0.063, "common_voice_15_en_test": 0.0723754, "earnings21_test": 0.07872434, "gigaspeech_test": 0.08227225, "librispeech_test_clean": 0.01628917, "librispeech_test_other": 0.03392333, "peoples_speech_test": 0.16221522, "tedlium3_long_form_test": 0.02533427, "tedlium3_test": 0.02250317, "cna_test": 0.11657028, "idpc_short_test": 0.14932127, "idpc_test": 0.17231348, "mediacorp_short_test": 0.10851231, "mediacorp_test": 0.09783408, "parliament_short_test": 0.04797029, "parliament_test": 0.05049817, "ukusnews_short_test": 0.05122813, "ukusnews_test": 0.04855613, "earnings22_test": 0.11704241 }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_bloomspeech_en_30": 0.06080662, "asr_cv21_en_30": 0.06417785, "asr_esd_en_30": 0.03473434, "average": 0.0669, "common_voice_15_en_test": 0.07112718, "gigaspeech_test": 0.08727358, "librispeech_test_clean": 0.01613817, "librispeech_test_other": 0.03091339, "peoples_speech_test": 0.15318691, "tedlium3_test": 0.02616566, "cna_test": 0.13662371, "idpc_short_test": 0.17020536, "idpc_test": 0.96338751, "mediacorp_short_test": 0.11278534, "mediacorp_test": 0.37490805, "parliament_short_test": 0.04771418, "parliament_test": 0.05355297, "ukusnews_short_test": 0.05598984, "ukusnews_test": 0.96558904, "earnings21_test": 0.09894478, "earnings22_test": 0.12255968, "tedlium3_long_form_test": 0.03627727 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 0.0733, "librispeech_test_clean": 0.02229143, "librispeech_test_other": 0.04134406, "common_voice_15_en_test": 0.076, "peoples_speech_test": 0.196, "gigaspeech_test": 0.088, "earnings21_test": 0.092, "earnings22_test": 0.128, "tedlium3_test": 0.02968728, "tedlium3_long_form_test": 0.035, "asr_bloomspeech_en_30": 0.06452947, "asr_cv21_en_30": 0.06879195, "asr_esd_en_30": 0.03790642, "parliament_short_test": 0.05429632, "ukusnews_short_test": 0.07142574, "idpc_short_test": 0.1418378, "mediacorp_short_test": 0.11323513, "mediacorp_test": 0.10527176 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "average": 0.0752, "librispeech_test_clean": 0.019, "librispeech_test_other": 0.036, "common_voice_15_en_test": 0.098, "peoples_speech_test": 0.146, "gigaspeech_test": 0.096, "earnings21_test": 0.108, "earnings22_test": 0.141, "tedlium3_test": 0.038, "tedlium3_long_form_test": 0.046, "asr_bloomspeech_en_30": 0.0509, "asr_cv21_en_30": 0.0875, "asr_esd_en_30": 0.0362, "parliament_short_test": 0.05685747, "ukusnews_short_test": 0.05860879 }, { "model": "MERaLiON-3-3B-ASR-CTM", "asr_bloomspeech_en_30": 0.05997932, "asr_cv21_en_30": 0.07256711, "asr_esd_en_30": 0.0409199, "common_voice_15_en_test": 0.08381088, "earnings21_test": 0.10542526, "gigaspeech_test": 0.08832938, "librispeech_test_clean": 0.02144205, "librispeech_test_other": 0.04204449, "peoples_speech_test": 0.1971795, "tedlium3_long_form_test": 0.02976777, "tedlium3_test": 0.02919425, "average": 0.0757, "cna_test": 0.13195734, "idpc_short_test": 0.13609467, "idpc_test": 0.16632444, "mediacorp_short_test": 0.11031148, "mediacorp_test": 0.09693502, "parliament_short_test": 0.04712511, "parliament_test": 0.0499812, "ukusnews_short_test": 0.06305305, "ukusnews_test": 0.05824474, "earnings22_test": 0.13758519 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "asr_bloomspeech_en_30": 0.0655636, "asr_cv21_en_30": 0.10675336, "average": 0.0812, "earnings21_test": 0.10994724, "cna_test": 0.15721335, "earnings22_test": 0.15852595, "parliament_short_test": 0.0593674, "ukusnews_short_test": 0.06285465, "asr_esd_en_30": 0.03156225, "idpc_short_test": 0.1856944, "librispeech_test_clean": 0.01744054, "librispeech_test_other": 0.04299101, "mediacorp_short_test": 0.1288654, "mediacorp_test": 0.11679608, "tedlium3_test": 0.0336315, "idpc_test": 0.16273101, "parliament_test": 0.06102547, "tedlium3_long_form_test": 0.03634764, "ukusnews_test": 0.05798085, "common_voice_15_en_test": 0.11160849, "gigaspeech_test": 0.09415104, "peoples_speech_test": 0.16632158 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 0.0816, "librispeech_test_clean": 0.02463194, "librispeech_test_other": 0.0460956, "common_voice_15_en_test": 0.09, "peoples_speech_test": 0.205, "gigaspeech_test": 0.09, "earnings21_test": 0.108, "earnings22_test": 0.151, "tedlium3_test": 0.03215242, "tedlium3_long_form_test": 0.044, "asr_bloomspeech_en_30": 0.07218201, "asr_cv21_en_30": 0.07728607, "asr_esd_en_30": 0.03869944, "parliament_short_test": 0.0520169, "ukusnews_short_test": 0.07194159, "idpc_short_test": 0.15123564, "mediacorp_short_test": 0.11469695, "mediacorp_test": 0.11303637 }, { "model": "MERaLiON-3-3B-ASR", "asr_bloomspeech_en_30": 0.06204757, "asr_cv21_en_30": 0.07277685, "asr_esd_en_30": 0.03600317, "common_voice_15_en_test": 0.08520655, "earnings21_test": 0.1502058, "gigaspeech_test": 0.08962447, "librispeech_test_clean": 0.02253681, "librispeech_test_other": 0.04187411, "peoples_speech_test": 0.20414167, "tedlium3_long_form_test": 0.03518649, "tedlium3_test": 0.03204677, "average": 0.0859, "cna_test": 0.18942007, "idpc_short_test": 0.14340411, "idpc_test": 0.15793977, "mediacorp_short_test": 0.11154841, "mediacorp_test": 0.10224765, "parliament_short_test": 0.05783071, "parliament_test": 0.11004324, "ukusnews_short_test": 0.07634618, "ukusnews_test": 0.05549272, "earnings22_test": 0.19916403, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_bloomspeech_en_30": 0.05294726, "asr_cv21_en_30": 0.05536913, "asr_esd_en_30": 0.03076923, "average": 0.0878, "cna_test": 0.11736616, "common_voice_15_en_test": 0.06105569, "gigaspeech_test": 0.07929629, "idpc_short_test": 0.16759485, "idpc_test": 0.42967146, "librispeech_test_clean": 0.01296716, "librispeech_test_other": 0.02449598, "mediacorp_short_test": 0.10457663, "mediacorp_test": 0.11009399, "parliament_short_test": 0.04525547, "parliament_test": 0.07559451, "tedlium3_long_form_test": 0.19225897, "tedlium3_test": 0.02444006, "ukusnews_short_test": 0.05376771, "ukusnews_test": 0.20349845, "earnings21_test": 0.18000884, "earnings22_test": 0.18271728, "peoples_speech_test": 0.15729386 }, { "model": "MERaLiON-3-10B", "asr_bloomspeech_en_30": 0.05977249, "asr_esd_en_30": 0.04742268, "common_voice_15_en_test": 0.09374136, "gigaspeech_test": 0.1030213, "librispeech_test_clean": 0.03484334, "librispeech_test_other": 0.04268812, "tedlium3_test": 0.03003944, "average": 0.0879, "cna_test": 0.14421393, "idpc_short_test": 0.15175774, "idpc_test": 0.99469541, "mediacorp_short_test": 0.1119982, "mediacorp_test": 0.13870045, "parliament_short_test": 0.05614035, "parliament_test": 0.61608234, "ukusnews_short_test": 0.07059244, "ukusnews_test": 0.08169343, "earnings22_test": 0.18049401, "asr_cv21_en_30": 0.07361577, "earnings21_test": 0.13343278, "tedlium3_long_form_test": 0.06446165, "peoples_speech_test": 0.19139529, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "Fun-ASR-MLT-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512", "asr_bloomspeech_en_30": 0.06825233, "asr_cv21_en_30": 0.09144295, "asr_esd_en_30": 0.03679619, "average": 0.0883, "earnings21_test": 0.13742162, "cna_test": 0.14623017, "earnings22_test": 0.18630118, "parliament_short_test": 0.07683442, "ukusnews_short_test": 0.0969406, "idpc_short_test": 0.1943961, "librispeech_test_clean": 0.01745942, "librispeech_test_other": 0.04013251, "mediacorp_short_test": 0.16788485, "mediacorp_test": 0.16885983, "tedlium3_test": 0.04419637, "idpc_test": 0.19592745, "parliament_test": 0.07528903, "tedlium3_long_form_test": 0.05348346, "ukusnews_test": 0.12410465, "common_voice_15_en_test": 0.09585737, "gigaspeech_test": 0.10020316, "peoples_speech_test": 0.18783006 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 0.0896, "librispeech_test_clean": 0.01740279, "librispeech_test_other": 0.03880738, "common_voice_15_en_test": 0.079, "peoples_speech_test": 0.215, "gigaspeech_test": 0.099, "earnings21_test": 0.131, "earnings22_test": 0.226, "tedlium3_test": 0.04173123, "tedlium3_long_form_test": 0.051, "asr_bloomspeech_en_30": 0.0647363, "asr_cv21_en_30": 0.07927852, "asr_esd_en_30": 0.03251388, "parliament_short_test": 0.08305801, "idpc_short_test": 0.2841977, "mediacorp_short_test": 0.18576408, "mediacorp_test": 0.26138128, "ukusnews_short_test": 0.14547042 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 0.0947, "librispeech_test_clean": 0.01662892, "librispeech_test_other": 0.03854236, "common_voice_15_en_test": 0.088, "peoples_speech_test": 0.262, "gigaspeech_test": 0.114, "earnings21_test": 0.147, "earnings22_test": 0.197, "tedlium3_test": 0.03180025, "tedlium3_long_form_test": 0.071, "asr_bloomspeech_en_30": 0.05832472, "asr_cv21_en_30": 0.0747693, "asr_esd_en_30": 0.03790642, "parliament_short_test": 0.0578051, "ukusnews_short_test": 0.06539423, "idpc_short_test": 0.23947094, "mediacorp_short_test": 0.15259193, "mediacorp_test": 0.19869228 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 0.108, "librispeech_test_clean": 0.01570404, "librispeech_test_other": 0.03184098, "common_voice_15_en_test": 0.08, "peoples_speech_test": 0.312, "gigaspeech_test": 0.14, "earnings21_test": 0.189, "earnings22_test": 0.241, "tedlium3_test": 0.03218763, "tedlium3_long_form_test": 0.084, "asr_bloomspeech_en_30": 0.05873837, "asr_cv21_en_30": 0.07256711, "asr_esd_en_30": 0.03854084, "parliament_short_test": 0.0857216, "ukusnews_short_test": 0.18804809, "idpc_short_test": 0.24173338, "mediacorp_short_test": 0.18835039, "mediacorp_test": 0.41601962 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 0.1083, "librispeech_test_clean": 0.03303133, "librispeech_test_other": 0.05338381, "common_voice_15_en_test": 0.093, "peoples_speech_test": 0.206, "gigaspeech_test": 0.092, "earnings21_test": 0.219, "earnings22_test": 0.239, "tedlium3_test": 0.03835047, "tedlium3_long_form_test": 0.138, "asr_bloomspeech_en_30": 0.06246122, "asr_cv21_en_30": 0.08305369, "asr_esd_en_30": 0.04218874, "parliament_short_test": 0.06958637, "ukusnews_short_test": 0.08642514, "idpc_short_test": 0.32335538, "mediacorp_short_test": 0.13516249, "mediacorp_test": 0.16166735 }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2", "average": 0.1174, "librispeech_test_clean": 0.06243866, "librispeech_test_other": 0.07261713, "common_voice_15_en_test": 0.13013166, "peoples_speech_test": 0.21, "gigaspeech_test": 0.14723233, "earnings21_test": 0.16069169, "earnings22_test": 0.168, "tedlium3_test": 0.04662629, "tedlium3_long_form_test": 0.04451091, "asr_bloomspeech_en_30": 0.15305067, "asr_cv21_en_30": 0.12405621, "asr_esd_en_30": 0.088977, "cna_test": 0.19472595, "idpc_short_test": 0.18917508, "idpc_test": 0.18412047, "mediacorp_short_test": 0.12414258, "mediacorp_test": 0.11875766, "parliament_short_test": 0.06197977, "parliament_test": 0.06396278, "ukusnews_short_test": 0.09717868, "ukusnews_test": 0.10619769 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 0.1237, "librispeech_test_clean": 0.04898075, "librispeech_test_other": 0.08117369, "common_voice_15_en_test": 0.114, "peoples_speech_test": 0.217, "gigaspeech_test": 0.117, "earnings21_test": 0.189, "earnings22_test": 0.235, "tedlium3_test": 0.081596, "tedlium3_long_form_test": 0.087, "asr_bloomspeech_en_30": 0.1098242, "asr_cv21_en_30": 0.13611577, "asr_esd_en_30": 0.06724822, "idpc_short_test": 0.27793247, "mediacorp_short_test": 0.20915327, "mediacorp_test": 0.24527993, "parliament_short_test": 0.10628762, "ukusnews_short_test": 0.12785207, "cna_test": 0.19605242, "idpc_test": 0.35609172, "parliament_test": 0.20930069, "ukusnews_test": 0.23614567 }, { "model": "Omnilingual-LLM-ASR-7B [with language code]", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.132, "librispeech_test_clean": 0.027, "librispeech_test_other": 0.049, "common_voice_15_en_test": 0.125, "peoples_speech_test": 0.287, "gigaspeech_test": 0.164, "earnings21_test": 0.204, "earnings22_test": 0.258, "tedlium3_test": 0.101, "tedlium3_long_form_test": 0.144, "asr_bloomspeech_en_30": 0.07280248, "asr_cv21_en_30": 0.11377936, "asr_esd_en_30": 0.03806503, "parliament_short_test": 0.11087207, "ukusnews_short_test": 0.12963771 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 0.1362, "librispeech_test_clean": 0.11662892, "librispeech_test_other": 0.11784193, "common_voice_15_en_test": 0.077, "peoples_speech_test": 0.216, "gigaspeech_test": 0.145, "earnings21_test": 0.138, "earnings22_test": 0.166, "tedlium3_test": 0.07744048, "tedlium3_long_form_test": 0.105, "parliament_short_test": 0.05882956, "ukusnews_short_test": 0.10023412, "asr_bloomspeech_en_30": 0.18738366, "asr_cv21_en_30": 0.0741401, "asr_esd_en_30": 0.21411578, "idpc_short_test": 0.16898712, "mediacorp_short_test": 0.13212639, "mediacorp_test": 0.12227217 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 0.1521, "librispeech_test_clean": 0.02, "librispeech_test_other": 0.043, "common_voice_15_en_test": 0.113, "peoples_speech_test": 0.314, "gigaspeech_test": 0.13, "earnings21_test": 0.266, "earnings22_test": 0.366, "tedlium3_test": 0.041, "tedlium3_long_form_test": 0.291, "asr_bloomspeech_en_30": 0.08562565, "asr_cv21_en_30": 0.10927013, "asr_esd_en_30": 0.04631245 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 0.1553, "asr_bloomspeech_en_30": 0.05501551, "earnings21_test": 0.991, "asr_cv21_en_30": 0.08389262, "asr_esd_en_30": 0.06201427, "common_voice_15_en_test": 0.09868087, "gigaspeech_test": 0.09254031, "mediacorp_short_test": 0.11514674, "librispeech_test_clean": 0.0316157, "librispeech_test_other": 0.04719356, "peoples_speech_test": 0.16292794, "cna_test": 0.16119276, "tedlium3_test": 0.03352585, "idpc_short_test": 0.17786286, "mediacorp_test": 0.46653045, "parliament_short_test": 0.04973748, "ukusnews_short_test": 0.0602357, "idpc_test": 0.96851472, "parliament_test": 0.99478334, "ukusnews_test": 0.9537058, "earnings22_test": 0.15426326, "tedlium3_long_form_test": 0.05080929 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_bloomspeech_en_30": 0.29079628, "asr_cv21_en_30": 0.22483221, "asr_esd_en_30": 0.19920698, "average": 0.1643, "cna_test": 0.4554571, "parliament_short_test": 0.08405686, "ukusnews_short_test": 0.08638546, "idpc_short_test": 0.27271145, "librispeech_test_clean": 0.04709324, "earnings21_test": 0.14066738, "librispeech_test_other": 0.11146238, "mediacorp_short_test": 0.21185202, "mediacorp_test": 0.1770331, "tedlium3_test": 0.0640231, "earnings22_test": 0.19600943, "idpc_test": 0.23528405, "parliament_test": 0.11603534, "tedlium3_long_form_test": 0.04711471, "ukusnews_test": 0.07701877, "common_voice_15_en_test": 0.21754427, "gigaspeech_test": 0.14683905, "peoples_speech_test": 0.28584981 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 0.176, "librispeech_test_clean": 0.096, "librispeech_test_other": 0.118, "common_voice_15_en_test": 0.32, "peoples_speech_test": 0.242, "gigaspeech_test": 0.11, "earnings21_test": 0.277, "earnings22_test": 0.38, "tedlium3_test": 0.039, "tedlium3_long_form_test": 0.141, "asr_bloomspeech_en_30": 0.08273009, "asr_cv21_en_30": 0.25020973, "asr_esd_en_30": 0.05582871, "parliament_short_test": 0.08843642, "ukusnews_short_test": 0.08670291 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 0.189, "librispeech_test_clean": 0.050302, "librispeech_test_other": 0.10148604, "common_voice_15_en_test": 0.158, "peoples_speech_test": 0.375, "gigaspeech_test": 0.127, "earnings21_test": 0.379, "earnings22_test": 0.456, "tedlium3_test": 0.04623891, "tedlium3_long_form_test": 0.09, "asr_bloomspeech_en_30": 0.29141675, "asr_cv21_en_30": 0.13286493, "asr_esd_en_30": 0.06058684, "parliament_short_test": 0.12290946, "ukusnews_short_test": 0.11602714, "idpc_short_test": 0.32231117, "mediacorp_short_test": 0.19003711, "mediacorp_test": 0.25721291 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 0.269, "asr_cv21_en_30": 0.15384615, "asr_bloomspeech_en_30": 0.21344364, "earnings21_test": 0.99, "asr_esd_en_30": 0.18810468, "common_voice_15_en_test": 0.11950013, "gigaspeech_test": 0.232711, "mediacorp_short_test": 0.16248735, "librispeech_test_clean": 0.07746319, "librispeech_test_other": 0.08651207, "peoples_speech_test": 0.49304526, "cna_test": 1.97283387, "tedlium3_test": 0.09673898, "parliament_short_test": 0.08559355, "idpc_short_test": 0.22520014, "mediacorp_test": 0.48336739, "ukusnews_short_test": 0.14499425, "idpc_test": 0.97108145, "parliament_test": 0.99525331, "ukusnews_test": 0.95227324, "earnings22_test": 0.39324312, "tedlium3_long_form_test": 0.18314567 }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 0.2696, "librispeech_test_clean": 0.021, "librispeech_test_other": 0.048, "common_voice_15_en_test": 0.145, "peoples_speech_test": 0.379, "gigaspeech_test": 0.155, "earnings21_test": 0.645, "earnings22_test": 0.667, "tedlium3_test": 0.066, "tedlium3_long_form_test": 0.454, "asr_bloomspeech_en_30": 0.11292658, "asr_cv21_en_30": 0.48919883, "asr_esd_en_30": 0.05360825 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_bloomspeech_en_30": 0.52533609, "asr_cv21_en_30": 0.46277265, "asr_esd_en_30": 0.54274385, "cna_test": 1.0422879, "common_voice_15_en_test": 0.40672494, "earnings21_test": 0.17768017, "earnings22_test": 0.23722226, "gigaspeech_test": 0.24733529, "idpc_short_test": 0.3515489, "idpc_test": 0.31622177, "librispeech_test_clean": 0.10275576, "librispeech_test_other": 0.17792712, "mediacorp_short_test": 0.28955358, "mediacorp_test": 0.23269309, "parliament_short_test": 0.12470227, "parliament_test": 0.13278974, "peoples_speech_test": 0.47067393, "tedlium3_long_form_test": 0.08258269, "tedlium3_test": 0.15350754, "ukusnews_short_test": 0.16622356, "ukusnews_test": 0.10137224, "average": 0.2989, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "Omnilingual-LLM-ASR-7B", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.4618, "librispeech_test_clean": 0.03, "librispeech_test_other": 0.056, "common_voice_15_en_test": 0.127, "peoples_speech_test": 0.978, "gigaspeech_test": 0.835, "earnings21_test": 0.516, "earnings22_test": 0.544, "tedlium3_test": 1.128, "tedlium3_long_form_test": 0.994, "asr_bloomspeech_en_30": 0.09555326, "asr_cv21_en_30": 0.12982383, "asr_esd_en_30": 0.10816812, "ukusnews_short_test": 0.96587437 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_bloomspeech_en_30": 2.5795243, "asr_cv21_en_30": 1.42135067, "asr_esd_en_30": 1.50436162, "average": 0.8322, "cna_test": 3.04069613, "parliament_short_test": 0.12178256, "ukusnews_short_test": 0.14979564, "earnings21_test": 0.15572222, "earnings22_test": 0.20461058, "idpc_test": 0.30013689, "parliament_test": 0.10527305, "tedlium3_long_form_test": 0.0656228, "ukusnews_test": 0.10649928, "idpc_short_test": 0.48659937, "librispeech_test_clean": 0.22319743, "librispeech_test_other": 0.65982016, "mediacorp_short_test": 0.41796919, "mediacorp_test": 0.24887617, "tedlium3_test": 0.17234822, "common_voice_15_en_test": 1.82106495, "gigaspeech_test": 0.41874353, "peoples_speech_test": 0.75982499 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "asr_bloomspeech_en_30": 4.40455016, "asr_cv21_en_30": 1.73993289, "asr_esd_en_30": 5.03267248, "average": 1.4748, "idpc_short_test": 1.00539506, "earnings21_test": 0.12772299, "librispeech_test_clean": 0.51015478, "librispeech_test_other": 1.01843824, "mediacorp_short_test": 0.35972113, "mediacorp_test": 0.22991418, "parliament_short_test": 0.08807786, "tedlium3_test": 0.3756515, "ukusnews_short_test": 0.10483711, "earnings22_test": 0.17759396, "tedlium3_long_form_test": 0.04954258, "common_voice_15_en_test": 2.25146481, "gigaspeech_test": 0.7238445, "cna_test": 2.510373, "peoples_speech_test": 1.28590095, "idpc_test": 0.23887748, "parliament_test": 0.08948209, "ukusnews_test": 0.09911031 } ], "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. WavLLM's error rate on asr_cv21_en_30 (0.489) reflects output-language drift rather than transcription accuracy: on roughly 23% of Common Voice clips it answers in German or Chinese despite prompts specifying English, and those answers are usually correct translations of the audio. Its rate on the other English sets in this table is 0.05-0.11. The drift tracks the audio, not the prompt -- it is 21-36% across all ten prompt variants on that dataset and under 2% on the others.", "ascending": true } }, { "key": "ASR-Singlish", "title": "Task: Automatic Speech Recognition - Singlish", "taskName": "asr_singlish", "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "MNSC-PART1-ASR", "internal": "imda_part1_asr_test", "description": "Speech recognition test data from the IMDA NSC project, Part 1.", "hfLink": null, "stats": { "num_rows": 3000, "audio_length": { "min": 2.14, "max": 13.68, "mean": 5.94, "median": 5.74, "std": 1.95, "total_hours": 4.95, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 3000, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Can you help recognize the speech and transcribe it word for word?", "answer": "all good citizens should learn to change a light bulb", "audioFile": "examples/imda_part1_asr_test/example_0.wav" }, { "instruction": "Please transcribe.", "answer": "consignments that fail our inspections or tests are not allowed to be sold", "audioFile": "examples/imda_part1_asr_test/example_1.wav" } ] }, { "display": "MNSC-PART2-ASR", "internal": "imda_part2_asr_test", "description": "Speech recognition test data from the IMDA NSC project, Part 2.", "hfLink": null, "stats": { "num_rows": 3000, "audio_length": { "min": 1.86, "max": 14.85, "mean": 4.85, "median": 4.44, "std": 1.85, "total_hours": 4.04, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 3000, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Can you help recognize the speech and transcribe it word for word?", "answer": "Acra one K M Saraca Road", "audioFile": "examples/imda_part2_asr_test/example_0.wav" }, { "instruction": "Could you convert the speech into a text transcript for me?", "answer": "Seafood Paella Yaki Soba and Gobchang Gui", "audioFile": "examples/imda_part2_asr_test/example_1.wav" } ] }, { "display": "MNSC-PART3-ASR", "internal": "imda_part3_30s_asr_test", "description": "Speech recognition test data from the IMDA NSC project, Part 3.", "hfLink": null, "stats": { "num_rows": 1000, "audio_length": { "min": 16.44, "max": 29.99, "mean": 27.74, "median": 28.24, "std": 1.93, "total_hours": 7.7, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 1000, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Please transcribe.", "answer": ": I mean he they get paid [lah] so so (uh) usually usually I would return my own things I would return my own trays but for clear cleaning the table that's a different thing I will clear my own table ya and usually if I drop any food on the floor I will pick it up before ya and I will throw it away [lah] so honestly I feel that's the best I can do to help them", "audioFile": "examples/imda_part3_30s_asr_test/example_0.wav" }, { "instruction": "Please transcribe the content of this audio into text format.", "answer": ": but he actually like he ask me if he could go to the petrol station to pump some (uh) to pump the petrol then after that (mm) my daughter is already very tired already but okay [lah] to be nice so maybe it take takes maybe less than five minutes so I just proceed with t~ proceed ya so afterwards like okay I ask him to (uh) rush back home and then reach back home then I get my daughter to change all this and she's she slept she's so tired", "audioFile": "examples/imda_part3_30s_asr_test/example_1.wav" } ] }, { "display": "MNSC-PART4-ASR", "internal": "imda_part4_30s_asr_test", "description": "Speech recognition test data from the IMDA NSC project, Part 4.", "hfLink": null, "stats": { "num_rows": 1000, "audio_length": { "min": 15.13, "max": 30.0, "mean": 26.25, "median": 27.2, "std": 3.27, "total_hours": 7.29, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 1000, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Could you convert the speech into a text transcript for me?", "answer": ": okay this [one] we also do we get this customer and the other customer talk they can talk each other they ask ask the question [eh] [oh] ya how about coming in and ask you question all did they probably will answer all these question or not then I also answer them so this will give them give them the idea okay actually I can also take this approach why (err) I must bring #kumar# everytime here and and see his face I don't want to see #kumar# anymore so think about it it's very wise", "audioFile": "examples/imda_part4_30s_asr_test/example_0.wav" }, { "instruction": "Please transcribe the content of this audio into text format.", "answer": ": !wah! tadi kau okay (uh) everything okay breaking okay tak macam semalam cause it it was the same fella ya the same instructor\n: the same instructor\n: then I said [oh] then then why I cannot the same ride again [ah] [oh] takde apa-apa kau just need a a bit a bit a bit a of practice only just familiarise a bit more with the bike\n: I think they they they want more money [ah]\n: no then inside my head I'm thinking kepala otak kau kau pun every prac kau pegang motor ke sama juga [kan] it's the same it's the same model [what] it's the same bike [what] what what is it that's so different\n: it's the same bike", "audioFile": "examples/imda_part4_30s_asr_test/example_1.wav" } ] }, { "display": "MNSC-PART5-ASR", "internal": "imda_part5_30s_asr_test", "description": "Speech recognition test data from the IMDA NSC project, Part 5.", "hfLink": null, "stats": { "num_rows": 1000, "audio_length": { "min": 15.02, "max": 29.99, "mean": 24.87, "median": 25.88, "std": 3.92, "total_hours": 6.91, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 1000, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Please transcribe.", "answer": ": like my relatives right so a joint account has been quite (uh) a issue [lah] they has popped up especially with (uh) some my relatives cause a bit of dispute but I think joint account is good cause there are a few benefits just like itself and one is it okay let's let's talk about like in a touch wood situation right when something happens to one", "audioFile": "examples/imda_part5_30s_asr_test/example_0.wav" }, { "instruction": "Please transcribe the content of this audio into text format.", "answer": ": I realize right a lot of a lot of cosmetic products right actually they are being so rapidly rapidly like shopee lazada and all that and actually I wonder be because they do claim to be the official store you know they don't claim to like clarence I know I use clarence still right like face wash and all that they do use (uh) they do have official source", "audioFile": "examples/imda_part5_30s_asr_test/example_1.wav" } ] }, { "display": "MNSC-PART6-ASR", "internal": "imda_part6_30s_asr_test", "description": "Speech recognition test data from the IMDA NSC project, Part 6.", "hfLink": null, "stats": { "num_rows": 1000, "audio_length": { "min": 15.09, "max": 30.0, "mean": 24.33, "median": 25.09, "std": 3.91, "total_hours": 6.76, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 1000, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Can you help recognize the speech and transcribe it word for word?", "answer": ": [oh] okay because I have friends (um) whose children have also been awarded scholarships and what happen is that after they have been awarded scholarships they actually return back to become a teacher so (um) I was under the impression that (um) the scholarship actually allows the students", "audioFile": "examples/imda_part6_30s_asr_test/example_0.wav" }, { "instruction": "Please transcribe.", "answer": ": ya for yes for your new employer\n: three months [oh] four month new employer but next year I only have two days so assuming but can it can it be a scenario like next year I only have two days of paid childcare leave (um) I leave the company maybe somewhere in the month of july but for the first seven months of the year first six months of the year I already consume two days of childcare leave", "audioFile": "examples/imda_part6_30s_asr_test/example_1.wav" } ] }, { "display": "SG-Streets-Utterance-30 [SEA]", "internal": "asr_sg_streets_utterance_30", "description": "SG Streets Utterance ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} } ], "data": { "columns": [ { "key": "imda_part1_asr_test", "display": "MNSC-PART1-ASR", "description": "Speech recognition test data from the IMDA NSC project, Part 1.", "hfLink": null, "isWer": true }, { "key": "imda_part2_asr_test", "display": "MNSC-PART2-ASR", "description": "Speech recognition test data from the IMDA NSC project, Part 2.", "hfLink": null, "isWer": true }, { "key": "imda_part3_30s_asr_test", "display": "MNSC-PART3-ASR", "description": "Speech recognition test data from the IMDA NSC project, Part 3.", "hfLink": null, "isWer": true }, { "key": "imda_part4_30s_asr_test", "display": "MNSC-PART4-ASR", "description": "Speech recognition test data from the IMDA NSC project, Part 4.", "hfLink": null, "isWer": true }, { "key": "imda_part5_30s_asr_test", "display": "MNSC-PART5-ASR", "description": "Speech recognition test data from the IMDA NSC project, Part 5.", "hfLink": null, "isWer": true }, { "key": "imda_part6_30s_asr_test", "display": "MNSC-PART6-ASR", "description": "Speech recognition test data from the IMDA NSC project, Part 6.", "hfLink": null, "isWer": true }, { "key": "asr_sg_streets_utterance_30", "display": "SG-Streets-Utterance-30 [SEA]", "description": "SG Streets Utterance ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true } ], "rows": [ { "model": "MERaLiON-3-3B-ASR-CTM", "asr_sg_streets_utterance_30": 0.08522823, "imda_part1_asr_test": 0.04550776, "imda_part2_asr_test": 0.05018558, "imda_part3_30s_asr_test": 0.1849649, "imda_part4_30s_asr_test": 0.24788393, "imda_part5_30s_asr_test": 0.13702088, "imda_part6_30s_asr_test": 0.09855885, "average": 0.1213, "ytb_asr_batch1": 0.07755367, "ytb_asr_batch2": 0.09176811 }, { "model": "MERaLiON-3-3B-ASR", "asr_sg_streets_utterance_30": 0.09044972, "imda_part1_asr_test": 0.04502, "imda_part2_asr_test": 0.05010661, "imda_part3_30s_asr_test": 0.19060034, "imda_part4_30s_asr_test": 0.24982307, "imda_part5_30s_asr_test": 0.13815895, "imda_part6_30s_asr_test": 0.10039656, "average": 0.1235, "ytb_asr_batch1": 0.08514127, "ytb_asr_batch2": 0.09481696, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 0.1296, "imda_part1_asr_test": 0.044, "imda_part2_asr_test": 0.054, "imda_part3_30s_asr_test": 0.20149102, "imda_part4_30s_asr_test": 0.26390658, "imda_part5_30s_asr_test": 0.14425785, "imda_part6_30s_asr_test": 0.10284683, "asr_sg_streets_utterance_30": 0.09668183, "ytb_asr_batch1": 0.08045144, "ytb_asr_batch2": 0.10093628 }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2", "average": 0.1319, "imda_part1_asr_test": 0.05221442, "imda_part2_asr_test": 0.05476585, "imda_part3_30s_asr_test": 0.18677096, "imda_part4_30s_asr_test": 0.26878981, "imda_part5_30s_asr_test": 0.13702088, "imda_part6_30s_asr_test": 0.10025147, "asr_sg_streets_utterance_30": 0.1237999, "ytb_asr_batch1": 0.13329775, "ytb_asr_batch2": 0.13365191 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 0.1589, "imda_part1_asr_test": 0.043, "imda_part2_asr_test": 0.047, "imda_part3_30s_asr_test": 0.21839736, "imda_part4_30s_asr_test": 0.4006511, "imda_part5_30s_asr_test": 0.17129434, "imda_part6_30s_asr_test": 0.11753232, "asr_sg_streets_utterance_30": 0.11453596, "ytb_asr_batch1": 0.10054524, "ytb_asr_batch2": 0.13209505 }, { "model": "MERaLiON-3-10B", "imda_part1_asr_test": 0.05384841, "imda_part6_30s_asr_test": 0.10568398, "average": 0.1602, "ytb_asr_batch1": 0.09417776, "ytb_asr_batch2": 0.14792311, "asr_sg_streets_utterance_30": 0.09213407, "imda_part4_30s_asr_test": 0.32055202, "imda_part2_asr_test": 0.14885888, "imda_part3_30s_asr_test": 0.23093114, "imda_part5_30s_asr_test": 0.16928083, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 0.1623, "imda_part1_asr_test": 0.052, "imda_part2_asr_test": 0.145, "imda_part3_30s_asr_test": 0.2389973, "imda_part4_30s_asr_test": 0.31170559, "imda_part5_30s_asr_test": 0.16472854, "imda_part6_30s_asr_test": 0.12768804, "asr_sg_streets_utterance_30": 0.09567121, "ytb_asr_batch1": 0.09440653, "ytb_asr_batch2": 0.1089368 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 0.1846, "imda_part1_asr_test": 0.049, "imda_part2_asr_test": 0.058, "imda_part3_30s_asr_test": 0.28080825, "imda_part4_30s_asr_test": 0.44151451, "imda_part5_30s_asr_test": 0.20950727, "imda_part6_30s_asr_test": 0.16100848, "asr_sg_streets_utterance_30": 0.09213407, "ytb_asr_batch1": 0.12975178, "ytb_asr_batch2": 0.19218544 }, { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_sg_streets_utterance_30": 0.11840997, "average": 0.1966, "imda_part1_asr_test": 0.05662862, "imda_part2_asr_test": 0.24307036, "imda_part3_30s_asr_test": 0.24971144, "imda_part4_30s_asr_test": 0.41719745, "imda_part5_30s_asr_test": 0.16443673, "imda_part6_30s_asr_test": 0.12683367, "ytb_asr_batch1": 0.08331109, "ytb_asr_batch2": 0.0962657 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_sg_streets_utterance_30": 0.10089271, "average": 0.2163, "ytb_asr_batch1": 0.07740115, "imda_part1_asr_test": 0.05348259, "imda_part2_asr_test": 0.17408987, "imda_part3_30s_asr_test": 0.27051507, "imda_part4_30s_asr_test": 0.61643312, "imda_part5_30s_asr_test": 0.16999577, "imda_part6_30s_asr_test": 0.12855853, "ytb_asr_batch2": 0.08871927 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "asr_sg_streets_utterance_30": 0.09297625, "imda_part3_30s_asr_test": 0.29911327, "imda_part4_30s_asr_test": 0.67524416, "imda_part5_30s_asr_test": 0.17999037, "imda_part6_30s_asr_test": 0.15336751, "ytb_asr_batch1": 0.09478781, "ytb_asr_batch2": 0.12740286, "average": 0.2173, "imda_part1_asr_test": 0.0333626, "imda_part2_asr_test": 0.08702519 }, { "model": "Fun-ASR-MLT-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512", "asr_sg_streets_utterance_30": 0.10577733, "average": 0.2206, "ytb_asr_batch1": 0.16147482, "imda_part3_30s_asr_test": 0.32701892, "imda_part4_30s_asr_test": 0.68604388, "imda_part5_30s_asr_test": 0.18943053, "imda_part6_30s_asr_test": 0.15528581, "ytb_asr_batch2": 0.17919, "imda_part1_asr_test": 0.04302019, "imda_part2_asr_test": 0.03747137 }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_sg_streets_utterance_30": 0.14232777, "imda_part1_asr_test": 0.04421145, "imda_part2_asr_test": 0.34099389, "imda_part3_30s_asr_test": 0.27481973, "imda_part4_30s_asr_test": 0.49617834, "imda_part5_30s_asr_test": 0.1718196, "imda_part6_30s_asr_test": 0.13012219, "average": 0.2286, "ytb_asr_batch1": 0.08849659, "ytb_asr_batch2": 0.13763055 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "average": 0.232, "imda_part1_asr_test": 0.069, "imda_part2_asr_test": 0.319, "imda_part3_30s_asr_test": 0.267, "imda_part4_30s_asr_test": 0.457, "imda_part5_30s_asr_test": 0.211, "imda_part6_30s_asr_test": 0.17, "asr_sg_streets_utterance_30": 0.1312 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 0.2487, "imda_part3_30s_asr_test": 0.27363833, "imda_part6_30s_asr_test": 0.14756424, "ytb_asr_batch2": 0.09220058, "imda_part1_asr_test": 0.07657789, "ytb_asr_batch1": 0.08895413, "imda_part5_30s_asr_test": 0.19142945, "imda_part2_asr_test": 0.35473427, "imda_part4_30s_asr_test": 0.55750885, "asr_sg_streets_utterance_30": 0.13946438 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 0.2862, "imda_part1_asr_test": 0.072, "imda_part2_asr_test": 0.191, "imda_part3_30s_asr_test": 0.38160807, "imda_part4_30s_asr_test": 0.59007785, "imda_part5_30s_asr_test": 0.28701285, "imda_part6_30s_asr_test": 0.24646162, "asr_sg_streets_utterance_30": 0.23496716, "ytb_asr_batch1": 0.19743013, "ytb_asr_batch2": 0.23050144 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 0.3138, "imda_part3_30s_asr_test": 0.29340992, "imda_part6_30s_asr_test": 0.18399587, "imda_part1_asr_test": 0.11091601, "ytb_asr_batch1": 0.11430968, "ytb_asr_batch2": 0.14909075, "imda_part5_30s_asr_test": 0.24149, "imda_part2_asr_test": 0.5233357, "imda_part4_30s_asr_test": 0.53639066, "asr_sg_streets_utterance_30": 0.30739431 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 0.3325, "imda_part1_asr_test": 0.053, "imda_part2_asr_test": 0.095, "imda_part3_30s_asr_test": 0.52678535, "imda_part4_30s_asr_test": 1.08227884, "imda_part5_30s_asr_test": 0.28049083, "imda_part6_30s_asr_test": 0.19881033, "asr_sg_streets_utterance_30": 0.09112346, "ytb_asr_batch1": 0.15171388, "ytb_asr_batch2": 0.30376024 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 0.3392, "imda_part1_asr_test": 0.053, "imda_part2_asr_test": 0.094, "imda_part3_30s_asr_test": 0.55255904, "imda_part4_30s_asr_test": 0.67276716, "imda_part5_30s_asr_test": 0.44870654, "imda_part6_30s_asr_test": 0.46588967, "asr_sg_streets_utterance_30": 0.08758632, "ytb_asr_batch1": 0.32988905, "ytb_asr_batch2": 0.41591887 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 0.3483, "imda_part1_asr_test": 0.058, "imda_part2_asr_test": 0.345, "imda_part3_30s_asr_test": 0.47432816, "imda_part4_30s_asr_test": 0.82431706, "imda_part5_30s_asr_test": 0.2961612, "imda_part6_30s_asr_test": 0.26125995, "asr_sg_streets_utterance_30": 0.17887822, "ytb_asr_batch1": 0.22141305, "ytb_asr_batch2": 0.3113283 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_sg_streets_utterance_30": 0.36921004, "average": 0.3869, "imda_part1_asr_test": 0.17939713, "ytb_asr_batch1": 0.16673657, "ytb_asr_batch2": 0.17211927, "imda_part3_30s_asr_test": 0.45879334, "imda_part4_30s_asr_test": 0.5722293, "imda_part5_30s_asr_test": 0.27586559, "imda_part6_30s_asr_test": 0.22055647, "imda_part2_asr_test": 0.63259101 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 0.4064, "imda_part1_asr_test": 0.129, "imda_part2_asr_test": 0.29, "imda_part3_30s_asr_test": 0.74535924, "imda_part4_30s_asr_test": 0.68356688, "imda_part5_30s_asr_test": 0.3934663, "imda_part6_30s_asr_test": 0.44280556, "asr_sg_streets_utterance_30": 0.16068722, "ytb_asr_batch1": 0.26564228, "ytb_asr_batch2": 0.4095617 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 0.407, "imda_part1_asr_test": 0.093, "imda_part2_asr_test": 0.458, "imda_part3_30s_asr_test": 0.681, "imda_part4_30s_asr_test": 0.787, "imda_part5_30s_asr_test": 0.375, "imda_part6_30s_asr_test": 0.255, "asr_sg_streets_utterance_30": 0.2002695 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 0.468, "imda_part1_asr_test": 0.106, "imda_part2_asr_test": 0.455, "imda_part3_30s_asr_test": 0.641, "imda_part4_30s_asr_test": 1.173, "imda_part5_30s_asr_test": 0.302, "imda_part6_30s_asr_test": 0.314, "asr_sg_streets_utterance_30": 0.28499242 }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 0.4965, "imda_part1_asr_test": 0.101, "imda_part2_asr_test": 0.446, "imda_part3_30s_asr_test": 0.754, "imda_part4_30s_asr_test": 1.144, "imda_part5_30s_asr_test": 0.398, "imda_part6_30s_asr_test": 0.425, "asr_sg_streets_utterance_30": 0.20717534 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_sg_streets_utterance_30": 0.91477177, "imda_part1_asr_test": 0.31777388, "imda_part2_asr_test": 1.44254916, "imda_part3_30s_asr_test": 0.49251097, "imda_part4_30s_asr_test": 0.63266808, "imda_part5_30s_asr_test": 0.33781753, "imda_part6_30s_asr_test": 0.27127059, "ytb_asr_batch1": 0.21092767, "ytb_asr_batch2": 0.21577616, "average": 0.6299, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "Omnilingual-LLM-ASR-7B [with language code]", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.6445, "imda_part1_asr_test": 0.182, "imda_part2_asr_test": 0.433, "imda_part3_30s_asr_test": 0.976, "imda_part4_30s_asr_test": 0.885, "imda_part5_30s_asr_test": 0.904, "imda_part6_30s_asr_test": 0.896, "asr_sg_streets_utterance_30": 0.2356409 }, { "model": "Omnilingual-LLM-ASR-7B", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.6445, "imda_part1_asr_test": 0.182, "imda_part2_asr_test": 0.433, "imda_part3_30s_asr_test": 0.976, "imda_part4_30s_asr_test": 0.885, "imda_part5_30s_asr_test": 0.904, "imda_part6_30s_asr_test": 0.896, "asr_sg_streets_utterance_30": 0.2356409 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "imda_part3_30s_asr_test": 0.70346682, "average": 1.5705, "imda_part4_30s_asr_test": 1.08086341, "imda_part5_30s_asr_test": 0.71014197, "imda_part6_30s_asr_test": 0.54225102, "ytb_asr_batch1": 0.34712319, "ytb_asr_batch2": 0.16541614, "imda_part1_asr_test": 1.40300946, "imda_part2_asr_test": 4.77197347, "asr_sg_streets_utterance_30": 1.78170793 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_sg_streets_utterance_30": 4.54067711, "average": 1.8782, "imda_part3_30s_asr_test": 0.586847, "imda_part4_30s_asr_test": 0.85224345, "imda_part5_30s_asr_test": 0.39212396, "imda_part6_30s_asr_test": 0.25821324, "ytb_asr_batch1": 0.20528463, "ytb_asr_batch2": 0.20606742, "imda_part1_asr_test": 1.34316164, "imda_part2_asr_test": 5.17381347 } ], "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "ASR-Mandarin", "title": "Task: Automatic Speech Recognition - Mandarin", "taskName": "asr_mandarin", "metric": "cer", "metricInfo": "Character Error Rate (CER) - the lower, the better. CER is used for every column in this table. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. MERaLiON-SpeechEncoder2-ASR-CTC returns an empty transcript on 3.4% of clips in commonvoice_zh_asr, and those count as full deletions in the error rate. This is the model's own output, not a failed run: two independent evaluations produced byte-identical predictions, and the clips it answers are transcribed normally. Scored over only the clips it does answer, the error rate is roughly 3-5 points lower.", "ascending": true, "datasets": [ { "display": "AISHELL-ASR-ZH", "internal": "aishell_asr_zh_test", "description": "ASR test dataset for Mandarin Chinese, based on the Aishell dataset.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/aishell_asr_zh_test_v1", "stats": { "num_rows": 6920, "audio_length": { "min": 1.86, "max": 14.7, "mean": 5.03, "median": 4.75, "std": 1.63, "total_hours": 9.67, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 6920, 0, 0, 0, 0 ] } } } }, { "display": "CommonVoice-ZH", "internal": "commonvoice_zh_asr", "description": "Speech Recognition dataset", "hfLink": null, "stats": { "num_rows": 5882, "audio_length": { "min": 1.01, "max": 10.97, "mean": 5.18, "median": 4.97, "std": 1.94, "total_hours": 8.47, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 5882, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Write down the spoken content in Chinese.", "answer": "阿契美尼斯的表现表明安善和波斯已经具有相当的独立性。", "audioFile": "examples/commonvoice_zh_asr/example_0.wav" }, { "instruction": "Transcribe the audio into text in Chinese.", "answer": "又可细分为城市地理学和乡村地理学。", "audioFile": "examples/commonvoice_zh_asr/example_1.wav" } ] }, { "display": "YouTube ASR: Chinese with English Prompt", "internal": "ytb_asr_batch3_chinese", "description": "YouTube Evaluation Dataset for ASR Task: This dataset contains Chinese and some Chinese-English codeswitch audio clips, featuring with English prompts. It includes approximately 3.32 hours of audio, with individual clips ranging from 17 seconds to 1966 seconds in length.", "hfLink": null, "stats": { "num_rows": 206, "audio_length": { "min": 17.0, "max": 1966.0, "mean": 58.04, "median": 34.0, "std": 148.81, "total_hours": 3.32, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 70, 125, 7, 3, 1 ] } } }, "examples": [ { "instruction": "Transcribe the input audio file into text.", "answer": ": (jingle)我个人觉得有些(啊)反馈是合理的。因为你可以考虑一下,就是送餐员在新加坡,是经常下雨,或者是大热天的时候,他们其实是under很大的工作压力的。所以他们会觉得说,每一次做一个订单,他会考量,真的是每一个订单,做完会考量说这个订单是值不值得。那以我们平台来说呢,我们的责任,就是就是make sure它每一个订单,都是值得的。那我们会做一些(啊),每个礼拜或每个月,会做一些调整(door knock)。是按照(啊)市场价格,给他们一个稳定的收入。", "audioFile": "examples/ytb_asr_batch3_chinese/example_0.wav" }, { "instruction": "What is the text spoken in this audio?", "answer": ": 大家好,我是六月。我们今天(呢)请到了一位很特别的来宾,我们欢迎Ryan。 : 哈啰,大家好,我是Ryan廖永宜。 : 欢迎,欢迎Ryan哥!我觉得大家应该会很好奇,说今天为什么我们会请到 Ryan哥来上节目。 : 我也很希望有一天,我是为了得奖而上这个节目。但是很不幸的,因为这个社会新闻(额)上了(上了)各大媒体的报导几个月。 : 你的那个袭击案件,是在2024年的11月22号。这个伤其实是蛮严重的。但是因为我现在看你,我不会觉得它是一个太严重的伤。 : 出事了之后,第一次化全妆。所以我觉得(ay)又有自信了。 : 这里。 : 一、二、三。 : (哦)还有耳朵。 : 四。(哦)她还帮我割了一个双眼皮(啦)。就是这边。 : 有,有,有。我看到了。 : 对,对,对。我很庆幸,就是我(我我)生存下来了。但是我很心痛的就是,(额)很多网民的这个军团(呢),就一直在attack attack 我。我用了几个小时看完所有人的comment。 : 你去看所有的comment(啊)?Comment 不是应该看的东西(啊)! : 对,我学到了, 太迟了! : 哈哈哈哈。 : 我看完comment之后,我忘记我是victim。之后我(我)朋友开导了我,还有一些热心的网民,跟我讲加油。走过街上,也有陌生人跟你说加油。我发现到就是,其实这些军团的网民,他是为了攻击你而攻击你。所以我学会了,就是不理会他们。有一则是500多个comment。 : (WOW) : (然后)我想95%都在骂我。 : 他们为什么要骂你? : 我不知道。我就觉得很奇怪。很无理的一个评论,讲说无风不起浪(啊)才会发生这件事情。我其实是一个不出去玩的人。 : (额)。 : 我蛮早一点。。。 : 虽然你看起来不像。 : 对,对,对。 : 我觉得有很大一部分,可能是因为因为我跟你合作过《不可饶恕的罪恶》。 : 对,对,对。 : 你演的那些强奸犯。 : 是,是,是。 : 可能就是因为你的很多戏剧作品,可能也因为是你演得太好。 : 谢谢。 : 或者是你太常演这些角色,(然后)对大家来说会有一个既定的印象。 : 有人真的是这样说活该。因为平时演坏人,因为平时都在你的社交媒体做一些(额)视频。但是我的视频永远有一个message的,就是告诉大家不要去做坏蛋。我所演过的角色,都不是我能选的。能够做刘德华,谁想做廖永宜? : 你要做你自己。 : (eh)(hey) : 你做廖永谊。 : (啊)对(umm)。经过这件事情了之后,我会更加小心。以前我(我)都很friendly的。来者不拒,谁要靠近我拍照(啊)。现在这件事情发生了之后,陌生人靠近我,我都会(额)怕。 : 可能有一点 PTSD。 : (哦)(耶)我从小到大,都有焦虑忧郁的这些问题 : 大慨 : 都在吃药。 : 几岁开始的? : 应该从中学开始吧。 : 哇。 : 对。 : 那你的这个忧郁症,其实发作的时间蛮早的。 : 是,是。但是那时候,还是可以控制。可是到了今年二月,拍完今年的这个贺岁片之后,我出现了很严重的幻听。 : 这是之前都没有的吗? : 之前都没有。一开始的时候是动物的声音。 : 动物的 : (Hmm) : 声音?对不起啊? : OK,没事,没事。 : 因为我好奇。我觉得很多人,现代人,都会觉得多多少少,有一些忧郁症,或者是一些心理上的疾病。很多时候,我们是不知道去怎么理清这个东西,到底是什么?所以在下我才会想要继续问下去。所以那个时候你听到的这个动物的声音,是真实的吗? : 近年来,就一直在我头脑里面。否定我的工作,否定我的everything,可是我还能够克制。 : (额) : 因为我吃药。 : (额) : 那么,今年它已经变成是一个,本来是里面的声音,它变成是一个外面进来的声音。 : OK,所以你听到一些可能像什么样的声音? : (额)一个很可爱的鸟。它们平时如果,(laughing)sorry 啊,平时如果我做一些错的事情,它不要我做,它就会 “对对对对对” 。那如果是真的是做对的事情,它的那个 “对对对对对” 会更大声。OK。Sorry,这个我有跟医生说过也是。 : 你OK to分享这个东西吗? : OK, OK。因为我的出发点是希望说,所有的朋友们,如果你们有发现到自己的朋友,有这样的情况。其实你们的关怀,可以帮到她的。那么我愿意分享的原因,也是就是希望更多人可以得到帮助。这些年来,我自己都有那个,那把声音在嘲笑我。所以我很累了。这把声音,真的是,我想跟她断绝关系。 : (额) : 其实轻生是一件很恐怖的事情。你现在问我啊敢不敢?我不敢!它其实是在一个很冲动的,一个impulsive的决定,一刹那的决定。很多不幸的人,没有得到帮助的人,就因为那一刹那的(的)打击就。。。(额) : 你现在的状态你觉得。。。 : 它是很奇怪的。这个病是现在我很energetic,but我如果不吃药的话,那个evil thoughts就会来了。那这个药又会让我发福,所以还蛮不开心的。 : OK,我的理解啦,因为我身边,其实也是有一些忧郁症的朋友,就会觉得说忧郁症,其实是真的是你身体生病了。它不是你的错,它just生病了。 : 我觉得你很棒!做你的朋友的人,都会了解忧郁症的(的)这些想法。 : 那你觉得就是,如果像我是你朋友的话,那你觉得我做什么,会让你感觉比较好呢? : 买冰淇淋给我吃! : 真的吗? : 哈哈哈! : 哈哈哈! : 我的朋友都是这样这样哄我的。 : 真的哦! : 只要我一不开心,他们就会去(去)买冰淇淋给我吃。那个开锁事件出院的那一天,我的好朋友就买了30杯冰淇淋,我就吃了30杯了,在一晚里面。 : 那也太多了! : 是,是。 : 那你发福不是因为 : 对! : 那个药,是因为冰淇淋! : 哈哈哈,开玩笑啦!我觉得物质上是改变不了什么(什么)东西。 : 但是一个关怀嘛! : 对,关怀,对。 : 让你觉得有人在意你嘛! 对,对,对。 有人在乎你嘛!你的女朋友 : 嘿嘿嘿。 : 嘿嘿嘿,你们在一起七年? : 八年。 : 八年,她这样子陪你下来,(然后)包括就是你生病的时候,她也是可能第一个发现说她觉得你病得很严重了。 : 对,她其实在今年觉得我没有救了。 : 为什么?你做了什么事情她这么觉得? : 我有买碳回家。我有做什么东西,都被他没收。这些怪怪的东西。(额)我女朋友在这8年里面,陪我经历了好多。从我最风光到我最落魄。(额)今年我赶走她很多次,因为我觉得我自己也没救。你知道我在心理卫生学院的时候,(额)我不愿意入院。他们觉得我的幻听很严重,他们要我入院。(然后)我女朋友就流着眼泪,没有表情流着眼泪讲说,去!我会等你的。你知道发生什么事吗?就是你的ward,你看得到楼下的,那个马路。那时候,发生在8月15。 : 中秋节! : 中秋节的时候。她买着月饼,(唉)就是看着那个床,(然后)(额)就是她不知道我在哪里,她就陪着我。(哇)我哭到哭到真的是(然后)(嘿嘿嘿)。 : (嘿嘿嘿)我都想哭了。 : 是是是,就傻傻的站在那边,那么希望大家,有遇到这样好的另外一半。 : 那个晚上,是你不懂,你就偶然的机会你就往下看? : 因为我很想回家。我尝试要开门,我跟护士讲我想回家,但是护士不让我回家。我就一直就很像小孩子这样,看着那个窗口,看着。我这样就像一把森林这么熟悉的。你知道,因为我们不能拿手机的。 : 什么都没有?在里面? : 不能。(然后)只有电视机,(然后)医生给你药吃,你就睡觉那种。那结果我就很闷的时候,我就看到她在楼下。(哇) 跟你说,我出院的时候,我真的抱着她说,我看到你这一幕。她就哭着跟我讲说,你看到了。(然后)我觉得这样的女生,在我什么都没有了,还愿意做这种傻傻的事情。(呀)我真的觉得我很幸运。 : 所以当你有时候,有一些很负能量的东西的时候,你就会想一想,你看这么好的女生都会愿意陪在你身边。 : 对,我很幸运。 : 那你要不要借这个节目,跟你这个陪了你八年,这么不离不弃,你赶了她这么多次,她都不走的,这么好的一个女生,说一些话? : 我想说的是,我很对不起你。(额)如果时间能够重来,我知道我以后这样的话,我会选择不认识你。因为你在你最青春的时候,给了我你所有的青春期。我们是在一个电影的首映里认识的。是在转角的时候,她看到我。她妈妈是我的粉丝,就跟我拍了一张照片。如果能够回到那一天的话,(sobbing)(额)我会拒绝你,不想跟你拍照。(然后)对不起(额)你的家乡的朋友们,还有你的家庭,让你们蒙羞了。但是接下来,我会努力把你照顾好(呀)。 : 是,要努力振作,不要辜负他(sobbing)。 : 我会的。 : 那Ryan哥就是接下来工作的方面,你有什么安排? : 其实现在(就)顺其自然(嘛),(就)很幸运的,就是接下来,我很多工作。一些在演艺圈的(额)朋友们,他们(额)有些当了监制了,就给我一些工作。比如说我(我)最近就是要导一个贺岁的这个网剧。 : (嗯) : 国辉大哥是监制,那他给我这个机会(啊啊)做联合导演,就是实习,就学怎么样导戏。(啊)如果有戏拍是bonus(啦)。 : (嗯) : 没有的话(嘞)我很喜欢,就是现在我的平台,就是召集一些跟我一样,就是在人生上有一些(额)不什么顺利的人,进来我的平台,(然后)一起用演戏来当成是hobby。不是商业化的,所以我觉得是一个很(很很很)棒的东西。 : Acting therapy。 : (呀)这一些小小的快乐,可以让他们带来一点点色彩。 : (嗯) : 我觉得很有意义。 : 那你有没有什么角色你想要挑战呢?来这边许愿。我们这也是一个许愿池。 : 那我(我)希望接下来,我可以,那演好人,是不可能了的啦。 : (额)没有,这种骗子可能前面是坏的,然后他后面的时候,有一个转折还是什么,就变好人了,也是可以啊。 : 承你贵言,承你贵言。我希望接下来可以演一些有意义的角色,(然后)就是比较内心的角色。不管是什么角色,我都会尽力去演。我最终的目标,是想要拿个奖,然后退休。一个奖就好,我就退休了。 : 我觉得想要拿奖,是一个很好的目标。但是我更希望你可以去热爱你做的事情。 : 谢谢,谢谢。 : 因为拿奖这个东西很多时候,不是我们可以决定的。 : 对。 : 拿奖不代表你好。 : 对。 压力。 : 我们希望2025年,这个事情会成真! : (呀呀呀) : OK。 : 对! : 谢谢Ryan哥。 : 谢谢六月。 : 好好照顾。", "audioFile": "examples/ytb_asr_batch3_chinese/example_1.wav" } ] }, { "display": "Cv21-ZH-30 [SEA]", "internal": "asr_cv21_zh_30", "description": "CV21 Chinese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Fleurs-ZH-30 [SEA]", "internal": "asr_fleurs_zh_30", "description": "Fleurs Chinese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "ESD-ZH-30 [SEA]", "internal": "asr_esd_zh_30", "description": "Fleurs Chinese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Sgpccsc-Utterance-30 [SEA]", "internal": "asr_sgpccsc_utterance_30", "description": "SG PCCSC Utterance ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} } ], "data": { "columns": [ { "key": "aishell_asr_zh_test", "display": "AISHELL-ASR-ZH", "description": "ASR test dataset for Mandarin Chinese, based on the Aishell dataset.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/aishell_asr_zh_test_v1", "isWer": true }, { "key": "commonvoice_zh_asr", "display": "CommonVoice-ZH", "description": "Speech Recognition dataset", "hfLink": null, "isWer": true }, { "key": "ytb_asr_batch3_chinese", "display": "YouTube ASR: Chinese with English Prompt", "description": "YouTube Evaluation Dataset for ASR Task: This dataset contains Chinese and some Chinese-English codeswitch audio clips, featuring with English prompts. It includes approximately 3.32 hours of audio, with individual clips ranging from 17 seconds to 1966 seconds in length.", "hfLink": null, "isWer": true }, { "key": "asr_cv21_zh_30", "display": "Cv21-ZH-30 [SEA]", "description": "CV21 Chinese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_fleurs_zh_30", "display": "Fleurs-ZH-30 [SEA]", "description": "Fleurs Chinese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_esd_zh_30", "display": "ESD-ZH-30 [SEA]", "description": "Fleurs Chinese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_sgpccsc_utterance_30", "display": "Sgpccsc-Utterance-30 [SEA]", "description": "SG PCCSC Utterance ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true } ], "rows": [ { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_cv21_zh_30": 0.0549, "asr_fleurs_zh_30": 0.0686, "asr_esd_zh_30": 0.016, "asr_sgpccsc_utterance_30": 0.0401, "average": 0.0637, "aishell_asr_zh_test": 0.0286, "commonvoice_zh_asr": 0.0879, "ytb_asr_batch3_chinese": 0.1499 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "asr_cv21_zh_30": 0.0632, "asr_fleurs_zh_30": 0.0667, "average": 0.0725, "aishell_asr_zh_test": 0.0353, "commonvoice_zh_asr": 0.1009, "asr_esd_zh_30": 0.0229, "asr_sgpccsc_utterance_30": 0.0483, "ytb_asr_batch3_chinese": 0.1703 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 0.0749, "asr_cv21_zh_30": 0.0489, "asr_fleurs_zh_30": 0.064, "asr_esd_zh_30": 0.0202, "asr_sgpccsc_utterance_30": 0.0452, "ytb_asr_batch3_chinese": 0.2365, "aishell_asr_zh_test": 0.0344, "commonvoice_zh_asr": 0.0753 }, { "model": "MERaLiON-3-3B-ASR-CTM", "asr_cv21_zh_30": 0.0796, "asr_fleurs_zh_30": 0.0834, "asr_esd_zh_30": 0.0303, "asr_sgpccsc_utterance_30": 0.0666, "aishell_asr_zh_test": 0.0507, "commonvoice_zh_asr": 0.135, "ytb_asr_batch3_chinese": 0.1297, "average": 0.0822 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 0.103, "aishell_asr_zh_test": 0.0453, "commonvoice_zh_asr": 0.1216, "ytb_asr_batch3_chinese": 0.2895, "asr_cv21_zh_30": 0.0819, "asr_fleurs_zh_30": 0.0867, "asr_esd_zh_30": 0.0302, "asr_sgpccsc_utterance_30": 0.0657 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 0.0858, "asr_cv21_zh_30": 0.0557, "asr_fleurs_zh_30": 0.0667, "asr_esd_zh_30": 0.0231, "asr_sgpccsc_utterance_30": 0.0454, "ytb_asr_batch3_chinese": 0.2808, "aishell_asr_zh_test": 0.0394, "commonvoice_zh_asr": 0.0897 }, { "model": "Fun-ASR-MLT-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512", "asr_cv21_zh_30": 0.0633, "asr_fleurs_zh_30": 0.0693, "asr_esd_zh_30": 0.0289, "average": 0.0869, "aishell_asr_zh_test": 0.0364, "commonvoice_zh_asr": 0.0976, "ytb_asr_batch3_chinese": 0.2627, "asr_sgpccsc_utterance_30": 0.0502 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_cv21_zh_30": 0.1004, "asr_fleurs_zh_30": 0.0671, "asr_esd_zh_30": 0.0137, "asr_sgpccsc_utterance_30": 0.0582, "average": 0.0924, "aishell_asr_zh_test": 0.0427, "commonvoice_zh_asr": 0.0946, "ytb_asr_batch3_chinese": 0.2702 }, { "model": "Omnilingual-LLM-ASR-7B [with language code]", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.1154, "aishell_asr_zh_test": 0.0767, "asr_cv21_zh_30": 0.1103, "asr_esd_zh_30": 0.0493, "asr_fleurs_zh_30": 0.0998, "asr_sgpccsc_utterance_30": 0.0932, "commonvoice_zh_asr": 0.1254, "ytb_asr_batch3_chinese": 0.2529 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 0.1222, "aishell_asr_zh_test": 0.063, "commonvoice_zh_asr": 0.1622, "asr_cv21_zh_30": 0.122, "asr_fleurs_zh_30": 0.1041, "asr_esd_zh_30": 0.0513, "asr_sgpccsc_utterance_30": 0.0692, "ytb_asr_batch3_chinese": 0.2835 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "average": 0.1256, "asr_cv21_zh_30": 0.1421, "asr_fleurs_zh_30": 0.0763, "asr_esd_zh_30": 0.0376, "asr_sgpccsc_utterance_30": 0.1625, "aishell_asr_zh_test": 0.0838, "commonvoice_zh_asr": 0.1791, "ytb_asr_batch3_chinese": 0.1975 }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2", "average": 0.1366, "aishell_asr_zh_test": 0.0766, "ytb_asr_batch3_chinese": 0.2022, "asr_cv21_zh_30": 0.1552, "asr_fleurs_zh_30": 0.1355, "asr_esd_zh_30": 0.0704, "asr_sgpccsc_utterance_30": 0.1386, "commonvoice_zh_asr": 0.1774 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 0.1312, "aishell_asr_zh_test": 0.0517, "commonvoice_zh_asr": 0.1437, "ytb_asr_batch3_chinese": 0.3926, "asr_cv21_zh_30": 0.1082, "asr_fleurs_zh_30": 0.1124, "asr_esd_zh_30": 0.0414, "asr_sgpccsc_utterance_30": 0.0681 }, { "model": "MERaLiON-3-10B", "asr_cv21_zh_30": 0.0955, "asr_fleurs_zh_30": 0.0908, "asr_esd_zh_30": 0.0348, "aishell_asr_zh_test": 0.0745, "average": 0.1415, "asr_sgpccsc_utterance_30": 0.1445, "commonvoice_zh_asr": 0.2083, "ytb_asr_batch3_chinese": 0.3423, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 0.1563, "asr_cv21_zh_30": 0.1316, "asr_fleurs_zh_30": 0.0923, "asr_esd_zh_30": 0.0516, "asr_sgpccsc_utterance_30": 0.1227, "ytb_asr_batch3_chinese": 0.4406, "aishell_asr_zh_test": 0.1025, "commonvoice_zh_asr": 0.1529 }, { "model": "MERaLiON-3-3B-ASR", "asr_cv21_zh_30": 0.087, "asr_fleurs_zh_30": 0.0841, "asr_esd_zh_30": 0.044, "asr_sgpccsc_utterance_30": 0.7652, "aishell_asr_zh_test": 0.0474, "commonvoice_zh_asr": 0.1283, "ytb_asr_batch3_chinese": 0.1386, "average": 0.1849, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 0.1985, "asr_cv21_zh_30": 0.0941, "asr_fleurs_zh_30": 0.1266, "asr_esd_zh_30": 0.1103, "asr_sgpccsc_utterance_30": 0.2968, "ytb_asr_batch3_chinese": 0.5254, "aishell_asr_zh_test": 0.1466, "commonvoice_zh_asr": 0.0896 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "average": 0.2244, "aishell_asr_zh_test": 0.1511, "asr_cv21_zh_30": 0.246, "asr_fleurs_zh_30": 0.1659, "asr_esd_zh_30": 0.2054, "asr_sgpccsc_utterance_30": 0.2067, "commonvoice_zh_asr": 0.2744, "ytb_asr_batch3_chinese": 0.321 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 0.2726, "asr_cv21_zh_30": 0.2928, "asr_fleurs_zh_30": 0.2129, "asr_esd_zh_30": 0.1887, "asr_sgpccsc_utterance_30": 0.3278, "ytb_asr_batch3_chinese": 0.4342, "aishell_asr_zh_test": 0.1245, "commonvoice_zh_asr": 0.3276 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_cv21_zh_30": 0.3907, "asr_fleurs_zh_30": 0.3455, "asr_esd_zh_30": 0.2692, "asr_sgpccsc_utterance_30": 0.3832, "average": 0.3621, "aishell_asr_zh_test": 0.2746, "commonvoice_zh_asr": 0.4174, "ytb_asr_batch3_chinese": 0.4541 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 0.4868, "aishell_asr_zh_test": 0.4743, "asr_cv21_zh_30": 0.4692, "asr_esd_zh_30": 0.3447, "commonvoice_zh_asr": 0.5327, "ytb_asr_batch3_chinese": 0.7152, "asr_fleurs_zh_30": 0.3784, "asr_sgpccsc_utterance_30": 0.4929 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 0.5223, "aishell_asr_zh_test": 0.4684, "asr_cv21_zh_30": 0.4653, "asr_esd_zh_30": 0.3248, "commonvoice_zh_asr": 0.508, "asr_fleurs_zh_30": 0.363, "asr_sgpccsc_utterance_30": 0.7874, "ytb_asr_batch3_chinese": 0.7391 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 0.5896, "asr_cv21_zh_30": 0.6702, "asr_esd_zh_30": 0.4805, "aishell_asr_zh_test": 0.6165, "commonvoice_zh_asr": 0.6713, "asr_fleurs_zh_30": 0.5358, "asr_sgpccsc_utterance_30": 0.5631 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_cv21_zh_30": 0.7664, "asr_fleurs_zh_30": 0.3805, "asr_esd_zh_30": 0.408, "asr_sgpccsc_utterance_30": 0.7681, "aishell_asr_zh_test": 0.7784, "commonvoice_zh_asr": 0.8463, "ytb_asr_batch3_chinese": 0.4883, "average": 0.6337, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "asr_cv21_zh_30": 0.7474, "asr_fleurs_zh_30": 0.2554, "asr_esd_zh_30": 1.079, "asr_sgpccsc_utterance_30": 2.4688, "average": 0.9011, "aishell_asr_zh_test": 0.4784, "ytb_asr_batch3_chinese": 0.3511, "commonvoice_zh_asr": 0.9278 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_cv21_zh_30": 1.0709, "asr_fleurs_zh_30": 0.5468, "asr_esd_zh_30": 1.2215, "asr_sgpccsc_utterance_30": 1.6661, "average": 1.0216, "aishell_asr_zh_test": 0.9517, "commonvoice_zh_asr": 1.1938, "ytb_asr_batch3_chinese": 0.5001 }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_cv21_zh_30": 3.0781, "asr_fleurs_zh_30": 3.0286, "asr_esd_zh_30": 3.036, "asr_sgpccsc_utterance_30": 2.6441, "aishell_asr_zh_test": 3.3046, "commonvoice_zh_asr": 2.9172, "average": 2.8732, "ytb_asr_batch3_chinese": 2.104 } ], "metric": "cer", "metricInfo": "Character Error Rate (CER) - the lower, the better. CER is used for every column in this table. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. MERaLiON-SpeechEncoder2-ASR-CTC returns an empty transcript on 3.4% of clips in commonvoice_zh_asr, and those count as full deletions in the error rate. This is the model's own output, not a failed run: two independent evaluations produced byte-identical predictions, and the clips it answers are transcribed normally. Scored over only the clips it does answer, the error rate is roughly 3-5 points lower.", "ascending": true } }, { "key": "ASR-Malay", "title": "Task: Automatic Speech Recognition - Malay", "taskName": "asr_malay", "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. MERaLiON-SpeechEncoder2-ASR-CTC returns an empty transcript on 5.1% of clips in asr_malcsc_30, and those count as full deletions in the error rate. This is the model's own output, not a failed run: two independent evaluations produced byte-identical predictions, and the clips it answers are transcribed normally. Scored over only the clips it does answer, the error rate is roughly 3-5 points lower.", "ascending": true, "datasets": [ { "display": "YouTube ASR: Malay with English Prompt", "internal": "ytb_asr_batch3_malay", "description": "YouTube Evaluation Dataset for ASR Task: This dataset mainly contains Malay and some Malay-English codeswitch audio clips, featuring with English prompts. It includes approximately 2.55 hours of audio, with indicidual clips ranging form 30 seconds to 95 seconds in length.", "hfLink": null, "stats": { "num_rows": 200, "audio_length": { "min": 30.0, "max": 95.0, "mean": 45.97, "median": 44.0, "std": 11.76, "total_hours": 2.55, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 0, 200, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Transcribe this audio and separate the text by speaker.", "answer": ": Mindef sedang memantau dengan teliti akan keadaan dan situasi (ah) kemanusiaan di Gaza terutama sekali pihak #civilian# yang sedang mempunyai #impact# oleh kerana situasi (ah) konflik di sana. : Seminggu lalu, hampir lima puluh pegawai Angkatan Bersenjata Singapura SAF bersama barangan bantuan kemanusiaan dikerahkan Singapura ke Gaza, sekaligus mencerminkan satu lagi usaha negara dalam membantu para mangsa yang terjerat dalam krisis kemanusiaan.", "audioFile": "examples/ytb_asr_batch3_malay/example_0.wav" }, { "instruction": "Transcribe the input audio file into text.", "answer": ": jadi ada yang cerita tu kadang-kadang kita nak buat promo tu macam nak : susah : susah : oh bermaksud cerita tu tak menarik : (ah) tak menarik : cuma ditanggungjawab sebagai #editor# kena #how# #to# #make# #it# #interesting# : (ah) kadang-kadang macam #let's# #say# dia tiga puluh minit tu kita tengok sekali apa cerita ni, kita sendiri yang kita tengok ni tak faham apa sebenarnya episod tu tau, kita macam tengok tengok-tengok nanti okay (lah), sampaikan macam satu team gitu pun macam eh apa cerita ni, asal cerita dia macam gini, tak tak namakan cerita mana (lah) #I# pun tak ingat tapi bila #part# dah lepas jadi promo tu, kita kasi #department# lain kasi tengok, (wah) cerita ni #best# (UNK) cerita ni #best# (eh) (UNK) (wah) korang (ppl) macam lepas tu (wah) kita (ah) tipu orang se macam tapi tu", "audioFile": "examples/ytb_asr_batch3_malay/example_1.wav" } ] }, { "display": "Fleurs-MS-30 [SEA]", "internal": "asr_fleurs_ms_30", "description": "Fleurs Malay ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Smaldusc-30 [SEA]", "internal": "asr_smaldusc_30", "description": "SMALDUSC Tagalog ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Malcsc-30 [SEA]", "internal": "asr_malcsc_30", "description": "MalCSC Burmese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} } ], "data": { "columns": [ { "key": "ytb_asr_batch3_malay", "display": "YouTube ASR: Malay with English Prompt", "description": "YouTube Evaluation Dataset for ASR Task: This dataset mainly contains Malay and some Malay-English codeswitch audio clips, featuring with English prompts. It includes approximately 2.55 hours of audio, with indicidual clips ranging form 30 seconds to 95 seconds in length.", "hfLink": null, "isWer": true }, { "key": "asr_fleurs_ms_30", "display": "Fleurs-MS-30 [SEA]", "description": "Fleurs Malay ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_smaldusc_30", "display": "Smaldusc-30 [SEA]", "description": "SMALDUSC Tagalog ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_malcsc_30", "display": "Malcsc-30 [SEA]", "description": "MalCSC Burmese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true } ], "rows": [ { "model": "MERaLiON-3-3B-ASR-CTM", "asr_fleurs_ms_30": 0.07066323, "asr_smaldusc_30": 0.04249726, "asr_malcsc_30": 0.2249101, "ytb_asr_batch3_malay": 0.16283178, "average": 0.1252 }, { "model": "MERaLiON-3-3B-ASR", "asr_fleurs_ms_30": 0.07568154, "asr_smaldusc_30": 0.04742607, "asr_malcsc_30": 0.23962079, "ytb_asr_batch3_malay": 0.15748709, "average": 0.1301, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2", "average": 0.1397, "ytb_asr_batch3_malay": 0.18235347, "asr_fleurs_ms_30": 0.09785705, "asr_smaldusc_30": 0.0456736, "asr_malcsc_30": 0.2327558 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 0.153, "ytb_asr_batch3_malay": 0.21025455, "asr_fleurs_ms_30": 0.10884308, "asr_smaldusc_30": 0.06210296, "asr_malcsc_30": 0.23095783 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 0.1575, "ytb_asr_batch3_malay": 0.2766102, "asr_fleurs_ms_30": 0.08036078, "asr_smaldusc_30": 0.04545455, "asr_malcsc_30": 0.22752534 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "average": 0.1682, "ytb_asr_batch3_malay": 0.22, "asr_fleurs_ms_30": 0.0764, "asr_smaldusc_30": 0.0577, "asr_malcsc_30": 0.3186 }, { "model": "MERaLiON-3-10B", "asr_fleurs_ms_30": 0.09568697, "asr_smaldusc_30": 0.04895947, "average": 0.1762, "asr_malcsc_30": 0.24321674, "ytb_asr_batch3_malay": 0.31674065, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "Fun-ASR-MLT-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512", "asr_fleurs_ms_30": 0.09100773, "asr_smaldusc_30": 0.03592552, "asr_malcsc_30": 0.35681595, "average": 0.2133, "ytb_asr_batch3_malay": 0.36932693 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 0.2249, "ytb_asr_batch3_malay": 0.49945647, "asr_fleurs_ms_30": 0.10423166, "asr_smaldusc_30": 0.05421687, "asr_malcsc_30": 0.24174567 }, { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_fleurs_ms_30": 0.1107419, "asr_smaldusc_30": 0.10427163, "asr_malcsc_30": 0.42530239, "average": 0.2312, "ytb_asr_batch3_malay": 0.28444605 }, { "model": "Omnilingual-LLM-ASR-7B [with language code]", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.2395, "ytb_asr_batch3_malay": 0.292, "asr_fleurs_ms_30": 0.08442968, "asr_malcsc_30": 0.50653808, "asr_smaldusc_30": 0.07513691 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_fleurs_ms_30": 0.10117998, "asr_smaldusc_30": 0.08598028, "asr_malcsc_30": 0.44589735, "average": 0.2403, "ytb_asr_batch3_malay": 0.3283359 }, { "model": "Omnilingual-LLM-ASR-7B", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.2621, "ytb_asr_batch3_malay": 0.324, "asr_fleurs_ms_30": 0.08626068, "asr_smaldusc_30": 0.0822563, "asr_malcsc_30": 0.55573717 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_fleurs_ms_30": 0.0908721, "asr_smaldusc_30": 0.08159912, "asr_malcsc_30": 0.74517816, "average": 0.291, "ytb_asr_batch3_malay": 0.24653501 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 0.4359, "ytb_asr_batch3_malay": 0.49211885, "asr_fleurs_ms_30": 0.31784891, "asr_smaldusc_30": 0.30087623, "asr_malcsc_30": 0.63288656 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 0.468, "ytb_asr_batch3_malay": 0.58682852, "asr_fleurs_ms_30": 0.16736742, "asr_smaldusc_30": 0.20668127, "asr_malcsc_30": 0.91124551 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 0.4979, "ytb_asr_batch3_malay": 0.67533291, "asr_fleurs_ms_30": 0.27932999, "asr_smaldusc_30": 0.2626506, "asr_malcsc_30": 0.77427264 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 0.6299, "ytb_asr_batch3_malay": 0.66926352, "asr_fleurs_ms_30": 0.44547674, "asr_smaldusc_30": 0.52201533, "asr_malcsc_30": 0.88296829 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_fleurs_ms_30": 0.14804015, "asr_smaldusc_30": 0.15410734, "asr_malcsc_30": 2.0040863, "ytb_asr_batch3_malay": 0.27751608, "average": 0.6459, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 0.6566, "ytb_asr_batch3_malay": 0.70649515, "asr_fleurs_ms_30": 0.23613183, "asr_smaldusc_30": 0.20503834, "asr_malcsc_30": 1.47891468 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "asr_fleurs_ms_30": 0.12552557, "asr_smaldusc_30": 0.18598028, "asr_malcsc_30": 2.01863354, "average": 0.6674, "ytb_asr_batch3_malay": 0.33952351 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 0.7464, "asr_fleurs_ms_30": 0.44771463, "ytb_asr_batch3_malay": 0.7406015, "asr_smaldusc_30": 0.55618839, "asr_malcsc_30": 1.24092841 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_fleurs_ms_30": 0.19462905, "asr_smaldusc_30": 0.25903614, "asr_malcsc_30": 2.33965348, "average": 0.7971, "ytb_asr_batch3_malay": 0.39500861 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 1.0, "ytb_asr_batch3_malay": 1.086, "asr_fleurs_ms_30": 0.96541435, "asr_smaldusc_30": 0.98444688, "asr_malcsc_30": 0.96404054 }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_fleurs_ms_30": 1.19130612, "asr_smaldusc_30": 1.24184009, "asr_malcsc_30": 1.23291925, "ytb_asr_batch3_malay": 1.00285352, "average": 1.1672 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "asr_fleurs_ms_30": 0.99138749, "asr_smaldusc_30": 1.24972618, "asr_malcsc_30": 1.69581563, "average": 1.2178, "ytb_asr_batch3_malay": 0.93441435 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "asr_fleurs_ms_30": 1.12457616, "asr_smaldusc_30": 1.29649507, "asr_malcsc_30": 1.31791435, "ytb_asr_batch3_malay": 1.41475677, "average": 1.2884 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 2.4753, "ytb_asr_batch3_malay": 0.99995471, "asr_fleurs_ms_30": 3.40092228, "asr_smaldusc_30": 2.22727273, "asr_malcsc_30": 3.27296502 } ], "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. MERaLiON-SpeechEncoder2-ASR-CTC returns an empty transcript on 5.1% of clips in asr_malcsc_30, and those count as full deletions in the error rate. This is the model's own output, not a failed run: two independent evaluations produced byte-identical predictions, and the clips it answers are transcribed normally. Scored over only the clips it does answer, the error rate is roughly 3-5 points lower.", "ascending": true } }, { "key": "ASR-Tamil", "title": "Task: Automatic Speech Recognition - Tamil", "taskName": "asr_tamil", "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. SALMONN-7B's scores in this table reflect an output-script limitation, not transcription accuracy: its Vicuna-7B backbone answers in Chinese rather than Tamil script (88-94% CJK predictions), so error rates exceed 1.0. The same model scores 0.06-0.25 on Latin-script cells. SeaLLMs-Audio-7B's Tamil scores reflect an output-script limitation rather than recognition accuracy: it renders Tamil audio in Latin, Thai or Chinese script and produces actual Tamil script on under 2% of predictions (asr_cv21_ta_30 and fleurs_tamil_ta_30_asr), so error rates exceed 1.0. This is specific to Tamil, not a general weakness -- the same model writes Thai script on 99.9% of asr_cv21_th_30 and scores 0.03-0.22 across the ASR-Thai table.", "ascending": true, "datasets": [ { "display": "CommonVoice-17-Tamil", "internal": "commonvoice_17_ta_asr", "description": "Speech Recognition dataset", "hfLink": null, "stats": { "num_rows": 12074, "audio_length": { "min": 1.62, "max": 10.66, "mean": 5.69, "median": 5.26, "std": 1.95, "total_hours": 19.09, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 12074, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Please transcribe.", "answer": "தாழச் சுடுவெய்யில் தாளாமல் நான்குளிர்ந்த", "audioFile": "examples/commonvoice_17_ta_asr/example_0.wav" }, { "instruction": "Please transcribe.", "answer": "சூதற்ற அகத்துறவு ஐம்புலனடக்க விளைவெனவும்", "audioFile": "examples/commonvoice_17_ta_asr/example_1.wav" } ] }, { "display": "Fleurs-Tamil", "internal": "fleurs_tamil_ta_30_asr", "description": "Tamil subset of the FLEURS multilingual benchmark, providing read speech for standardized ASR evaluation across languages and accents.", "hfLink": "https://arxiv.org/abs/2205.12446", "stats": { "num_rows": 705, "audio_length": { "min": 4.08, "max": 29.88, "mean": 13.04, "median": 12.28, "std": 4.56, "total_hours": 2.55, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 705, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Could you put the speech into writing in Tamil?", "answer": "அந்த நிறுவனம் தனது லாபம் ஈட்டும் வழிகளை பரவலாக்கியும் மற்றும் ஸ்கைப் பலமான நிலையில் உள்ள சீனா கிழக்கு ஐரோப்பா மற்றும் பிரேசிலில் அதிகப் புகழ் பெறவும் எண்ணுகிறது", "audioFile": "examples/fleurs_tamil_ta_30_asr/example_0.wav" }, { "instruction": "Write down what was said in Tamil, please.", "answer": "\"உயிர் தொடர்பான ஆய்விற்கு உயிரணுக்கள் அடிப்படையான மற்றும் இன்றியமையாதவை ஆகும் உண்மையில் அவற்றினை உயிரின் கட்டுமானத் தொகுதி' என்றும் அழைக்கப்படுகிறது.", "audioFile": "examples/fleurs_tamil_ta_30_asr/example_1.wav" } ] }, { "display": "YouTube ASR: Tamil with English Prompt", "internal": "ytb_asr_batch3_tamil", "description": "YouTube Evaluation Dataset for ASR Task: This dataset contains Tamil and some Tamil-English codeswitch audio clips, featuring with English prompts. It includes approximately 2.44 hours of audio, with individual clips ranging from 30 seconds to 324 seconds in length.", "hfLink": null, "stats": { "num_rows": 200, "audio_length": { "min": 30.0, "max": 331.0, "mean": 93.61, "median": 43.0, "std": 96.13, "total_hours": 5.2, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 0, 153, 24, 23, 0 ] } } }, "examples": [ { "instruction": "What is the text spoken in this audio?", "answer": ": உங்களுக்கு மசாலா தோசைய பத்தி சொல்லுனுனா, நீங்க லிட்டல் இந்தியாக்கு போனா, எதாச்சும் ஒரு popular இந்தியன் restaurant போறீங்கனு வச்சிக்கங்க. ஒரு மசாலா தோசை ஆறு வெள்ளி, ஆறு அம்பது, ஏழு (cashier ding) வெள்ளினு விக்கிறாங்க. (explosion) (music) கட்டுபடி ஆவுல. சரி, அதுனால cheapஆன மசாலா தோசலாம் எங்க கடைக்கும்? அப்படினு, நானு ஒரு தேடல ஆரம்பிச்சென். : (ah) : என்ன தெரிகிரது? : அப்புரோம்தான் தெரியவருது போன வருஷமே வீடமைப்பு வளர்ச்சி கழகம் வசதி குறைந்தவர்களுக்காகவும், முதியோர்களுக்காகவும், அவுங்களுடையா budget meal scheme, அதாவுது மலிவானா உணவு திட்டத்த அரிமுகபடுதிருக்காங்க. அதனால HDB கீழ இருக்குற அனைத்து உணவங்காடிகள்லையும் உணவு வித்தா, அது வந்து மூனு அம்பத்துக்கும் கீழ, ஏதாவது ஒரு உணவுப் பொருலாச்சும் மூனு அம்பத்துக்கு கீழ இருக்கணும் அப்படிங்கரத உருதி செஞ்சிருக்காங்க. பானம் விக்கிற கடையா இருந்தா அது குறைந்தது one twentyக்கும் கீழ இருக்கணும் அப்படிங்கரத உருதி செஞ்சிருக்காங்க. இதனால பல இந்திய கடைகள்ல (wind up) இந்த மசாலா தோச இருக்கு இல்லையா, இதுதான் வந்து (wind up) அவுங்க பட்ஜெட் மீல் விருப்பமா அவுங்க பரிமாறாங்க. இது மாதிரி அஞ்சு கடைக்கு (wind up) பொய் நான் சாப்பிட்டு அதோட விமர்சனத்த ஒரு கதையா எழுதியிருக்கேன். அந்த articleஅ படிக்க நீங்க தமிழ் முரசு செயலிய download பண்ணுங்க. (music) இப்ப என்ன…", "audioFile": "examples/ytb_asr_batch3_tamil/example_0.wav" }, { "instruction": "Transcribe the input audio file into text.", "answer": ": இவங்கல்லாம் போயிருப்பார்கள். எனக்கு அரவே இந்த duffel bag உள்ளுக்கு வாழ்ந்து பழக்கம் இல்ல. (ppl) எல்லாமே அது உள்ளுக்குதான் இருக்கு. நீங்க எடுக்கணும், போணும், வரணும். சில நேரம் கடுப்பாயிரும் அப்படியே தூக்கி அப்படியே உள்ளுக்கு போட்டுருவேன். இது என்னத்தா துரக்குறதுன்னு. (ppl) காலமிரை நீங்க திருப்பி pack பண்னணும், வந்துருவாங்க porters எடுத்துப் போர்த்துக்கு. they are very, timing எல்லாம் ரொம்ப correctடா இருக்கும். காலமிரை எடுத்துப் போவாங்க porters, அடுத்த townல கொண்டு போய் வச்சுருவாங்க. நம்ம சாய்ந்தரம் reach பன்னும்போது நம்மட்ட அங்க இருக்கும். whatever things you need to take out next morning it has to be pre pack again. ஒரு, ஒரு toothbrush எடுத்து நீங்க பல்ல வளக்க முடியாது. தண்ணிலா...", "audioFile": "examples/ytb_asr_batch3_tamil/example_1.wav" } ] }, { "display": "Cv21-TA-30 [SEA]", "internal": "asr_cv21_ta_30", "description": "CV21 Tamil ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Openslr-TA-30 [SEA]", "internal": "asr_openslr_ta_30", "description": "OpenSLR Tamil ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} } ], "data": { "columns": [ { "key": "commonvoice_17_ta_asr", "display": "CommonVoice-17-Tamil", "description": "Speech Recognition dataset", "hfLink": null, "isWer": true }, { "key": "fleurs_tamil_ta_30_asr", "display": "Fleurs-Tamil", "description": "Tamil subset of the FLEURS multilingual benchmark, providing read speech for standardized ASR evaluation across languages and accents.", "hfLink": "https://arxiv.org/abs/2205.12446", "isWer": true }, { "key": "ytb_asr_batch3_tamil", "display": "YouTube ASR: Tamil with English Prompt", "description": "YouTube Evaluation Dataset for ASR Task: This dataset contains Tamil and some Tamil-English codeswitch audio clips, featuring with English prompts. It includes approximately 2.44 hours of audio, with individual clips ranging from 30 seconds to 324 seconds in length.", "hfLink": null, "isWer": true }, { "key": "asr_cv21_ta_30", "display": "Cv21-TA-30 [SEA]", "description": "CV21 Tamil ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_openslr_ta_30", "display": "Openslr-TA-30 [SEA]", "description": "OpenSLR Tamil ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true } ], "rows": [ { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 0.3276, "commonvoice_17_ta_asr": 0.129, "fleurs_tamil_ta_30_asr": 0.26612032, "ytb_asr_batch3_tamil": 0.547, "asr_cv21_ta_30": 0.46821575, "asr_openslr_ta_30": 0.22780089 }, { "model": "MERaLiON-3-3B-ASR-CTM", "asr_cv21_ta_30": 0.45003612, "asr_openslr_ta_30": 0.21718548, "commonvoice_17_ta_asr": 0.29380405, "fleurs_tamil_ta_30_asr": 0.25891571, "ytb_asr_batch3_tamil": 0.63615244, "average": 0.3712 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 0.3761, "commonvoice_17_ta_asr": 0.156, "fleurs_tamil_ta_30_asr": 0.30313401, "ytb_asr_batch3_tamil": 0.664, "asr_cv21_ta_30": 0.49867566, "asr_openslr_ta_30": 0.25892985 }, { "model": "MERaLiON-3-3B-ASR", "asr_cv21_ta_30": 0.45918613, "asr_openslr_ta_30": 0.22234973, "commonvoice_17_ta_asr": 0.29753453, "fleurs_tamil_ta_30_asr": 0.25954611, "ytb_asr_batch3_tamil": 0.64462154, "average": 0.3766, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 0.3802, "commonvoice_17_ta_asr": 0.139, "fleurs_tamil_ta_30_asr": 0.28791427, "ytb_asr_batch3_tamil": 0.75, "asr_cv21_ta_30": 0.47724536, "asr_openslr_ta_30": 0.24673648 }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2", "average": 0.412, "commonvoice_17_ta_asr": 0.34470117, "fleurs_tamil_ta_30_asr": 0.31024856, "ytb_asr_batch3_tamil": 0.658, "asr_cv21_ta_30": 0.48880327, "asr_openslr_ta_30": 0.2582126 }, { "model": "MERaLiON-3-10B", "asr_cv21_ta_30": 0.47796773, "asr_openslr_ta_30": 0.23741214, "commonvoice_17_ta_asr": 0.33571705, "fleurs_tamil_ta_30_asr": 0.26413905, "average": 0.4587, "ytb_asr_batch3_tamil": 0.9780921, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "Omnilingual-LLM-ASR-7B", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.4731, "commonvoice_17_ta_asr": 0.314, "fleurs_tamil_ta_30_asr": 0.106, "ytb_asr_batch3_tamil": 0.868, "asr_cv21_ta_30": 0.47122562, "asr_openslr_ta_30": 0.60622579 }, { "model": "Omnilingual-LLM-ASR-7B [with language code]", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.4731, "commonvoice_17_ta_asr": 0.314, "fleurs_tamil_ta_30_asr": 0.106, "ytb_asr_batch3_tamil": 0.868, "asr_cv21_ta_30": 0.47122562, "asr_openslr_ta_30": 0.60622579 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "average": 0.4813, "commonvoice_17_ta_asr": 0.245, "fleurs_tamil_ta_30_asr": 0.231, "ytb_asr_batch3_tamil": 0.848, "asr_cv21_ta_30": 0.6197, "asr_openslr_ta_30": 0.4628 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_cv21_ta_30": 0.69022393, "asr_openslr_ta_30": 0.44197389, "average": 0.5983, "fleurs_tamil_ta_30_asr": 0.4474964, "ytb_asr_batch3_tamil": 0.8314121, "commonvoice_17_ta_asr": 0.58019879 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_cv21_ta_30": 0.70081869, "asr_openslr_ta_30": 0.5027973, "average": 0.666, "fleurs_tamil_ta_30_asr": 0.52107349, "commonvoice_17_ta_asr": 0.62748161, "ytb_asr_batch3_tamil": 0.97768041 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 0.6872, "commonvoice_17_ta_asr": 0.528, "fleurs_tamil_ta_30_asr": 0.68488833, "ytb_asr_batch3_tamil": 0.693, "asr_cv21_ta_30": 0.86588009, "asr_openslr_ta_30": 0.66446708 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_cv21_ta_30": 0.81025765, "asr_openslr_ta_30": 0.73777076, "commonvoice_17_ta_asr": 0.71166903, "fleurs_tamil_ta_30_asr": 0.54863112, "ytb_asr_batch3_tamil": 0.9109863, "average": 0.7439, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 0.9451, "asr_cv21_ta_30": 1.01962437, "fleurs_tamil_ta_30_asr": 0.8254683, "commonvoice_17_ta_asr": 1.07224732, "asr_openslr_ta_30": 0.85941759, "ytb_asr_batch3_tamil": 0.94865612 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_cv21_ta_30": 1.47279075, "asr_openslr_ta_30": 0.51742935, "average": 0.9695, "fleurs_tamil_ta_30_asr": 0.53881484, "ytb_asr_batch3_tamil": 1.1086867, "commonvoice_17_ta_asr": 1.20990061 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "asr_cv21_ta_30": 1.73886347, "asr_openslr_ta_30": 0.70176445, "average": 1.1106, "fleurs_tamil_ta_30_asr": 0.6126621, "ytb_asr_batch3_tamil": 0.95950715, "commonvoice_17_ta_asr": 1.5401833 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 1.1676, "asr_cv21_ta_30": 1.39898868, "fleurs_tamil_ta_30_asr": 1.10482709, "commonvoice_17_ta_asr": 1.32207306, "asr_openslr_ta_30": 1.02381294, "ytb_asr_batch3_tamil": 0.98826678 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 1.1952, "commonvoice_17_ta_asr": 0.847, "fleurs_tamil_ta_30_asr": 1.35572767, "ytb_asr_batch3_tamil": 1.362, "asr_cv21_ta_30": 1.19383578, "asr_openslr_ta_30": 1.21747239 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 1.1984, "commonvoice_17_ta_asr": 1.427, "fleurs_tamil_ta_30_asr": 1.508, "ytb_asr_batch3_tamil": 0.985, "asr_cv21_ta_30": 1.03575728, "asr_openslr_ta_30": 1.03629321 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "asr_cv21_ta_30": 1.24139176, "asr_openslr_ta_30": 1.28159518, "average": 1.2263, "fleurs_tamil_ta_30_asr": 1.28115994, "ytb_asr_batch3_tamil": 1.00764571, "commonvoice_17_ta_asr": 1.31951723 }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_cv21_ta_30": 1.32855767, "asr_openslr_ta_30": 1.37110888, "commonvoice_17_ta_asr": 1.36069446, "fleurs_tamil_ta_30_asr": 1.3454611, "average": 1.274, "ytb_asr_batch3_tamil": 0.96415338 }, { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_cv21_ta_30": 1.30050566, "asr_openslr_ta_30": 1.38043322, "average": 1.3355, "commonvoice_17_ta_asr": 1.3940364, "fleurs_tamil_ta_30_asr": 1.44614553, "ytb_asr_batch3_tamil": 1.15632535 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 1.3406, "commonvoice_17_ta_asr": 0.831, "fleurs_tamil_ta_30_asr": 1.57258646, "ytb_asr_batch3_tamil": 1.461, "asr_cv21_ta_30": 1.21562726, "asr_openslr_ta_30": 1.62300961 }, { "model": "Fun-ASR-MLT-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512", "asr_cv21_ta_30": 1.2688418, "asr_openslr_ta_30": 1.42705494, "average": 1.3771, "fleurs_tamil_ta_30_asr": 1.42011888, "ytb_asr_batch3_tamil": 1.44077516, "commonvoice_17_ta_asr": 1.3285788 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "fleurs_tamil_ta_30_asr": 1.48541066, "average": 1.4506, "asr_cv21_ta_30": 1.4128341, "asr_openslr_ta_30": 1.33266389, "ytb_asr_batch3_tamil": 1.51185085, "commonvoice_17_ta_asr": 1.51024913 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 1.7606, "commonvoice_17_ta_asr": 1.297, "fleurs_tamil_ta_30_asr": 1.16408501, "ytb_asr_batch3_tamil": 3.617, "asr_cv21_ta_30": 1.34023597, "asr_openslr_ta_30": 1.38488022 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 1.827, "commonvoice_17_ta_asr": 1.178, "fleurs_tamil_ta_30_asr": 1.35770893, "ytb_asr_batch3_tamil": 2.75, "asr_cv21_ta_30": 1.8302432, "asr_openslr_ta_30": 2.01907904 } ], "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. SALMONN-7B's scores in this table reflect an output-script limitation, not transcription accuracy: its Vicuna-7B backbone answers in Chinese rather than Tamil script (88-94% CJK predictions), so error rates exceed 1.0. The same model scores 0.06-0.25 on Latin-script cells. SeaLLMs-Audio-7B's Tamil scores reflect an output-script limitation rather than recognition accuracy: it renders Tamil audio in Latin, Thai or Chinese script and produces actual Tamil script on under 2% of predictions (asr_cv21_ta_30 and fleurs_tamil_ta_30_asr), so error rates exceed 1.0. This is specific to Tamil, not a general weakness -- the same model writes Thai script on 99.9% of asr_cv21_th_30 and scores 0.03-0.22 across the ASR-Thai table.", "ascending": true } }, { "key": "ASR-Indonesian", "title": "Task: Automatic Speech Recognition - Indonesian", "taskName": "asr_indonesian", "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "CommonVoice-17-Indonesian", "internal": "commonvoice_17_id_asr", "description": "Speech Recognition dataset", "hfLink": null, "stats": { "num_rows": 3641, "audio_length": { "min": 1.12, "max": 9.98, "mean": 4.11, "median": 3.89, "std": 1.26, "total_hours": 4.15, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 3641, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Please transcribe.", "answer": "Gadis itu datang menari ke arahku.", "audioFile": "examples/commonvoice_17_id_asr/example_0.wav" }, { "instruction": "Please transcribe.", "answer": "Di mana kamar kecilnya?", "audioFile": "examples/commonvoice_17_id_asr/example_1.wav" } ] }, { "display": "GigaSpeech-2-Indonesain", "internal": "gigaspeech2_id_test", "description": "Indonesian subset of the GigaSpeech 2 corpus, designed for evaluating ASR on real-world, long-form speech across diverse acoustic conditions.", "hfLink": "https://huggingface.co/datasets/speechcolab/gigaspeech2", "stats": { "num_rows": 5000, "audio_length": { "min": 0.15, "max": 16.47, "mean": 5.11, "median": 5.23, "std": 2.89, "total_hours": 7.1, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 5000, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Can you write out the speech in Indonesian?", "answer": "INILAH JAWABAN MENGAPA PENDUDUK OKINAWA SANGAT MEMPERHATIKAN DETAIL PEKERJAANNYA SEHARI HARI", "audioFile": "examples/gigaspeech2_id_test/example_0.wav" }, { "instruction": "Can you write out the speech in Indonesian?", "answer": "APAKAH SUDAH MUNCUL NAH DI SINI SUDAH MUNCUL YA TEMAN TEMAN", "audioFile": "examples/gigaspeech2_id_test/example_1.wav" } ] }, { "display": "Cv21-ID-30 [SEA]", "internal": "asr_cv21_id_30", "description": "CV21 Indonesian ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Fleurs-ID-30 [SEA]", "internal": "asr_fleurs_id_30", "description": "Fleurs Indonesian ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} } ], "data": { "columns": [ { "key": "commonvoice_17_id_asr", "display": "CommonVoice-17-Indonesian", "description": "Speech Recognition dataset", "hfLink": null, "isWer": true }, { "key": "gigaspeech2_id_test", "display": "GigaSpeech-2-Indonesain", "description": "Indonesian subset of the GigaSpeech 2 corpus, designed for evaluating ASR on real-world, long-form speech across diverse acoustic conditions.", "hfLink": "https://huggingface.co/datasets/speechcolab/gigaspeech2", "isWer": true }, { "key": "asr_cv21_id_30", "display": "Cv21-ID-30 [SEA]", "description": "CV21 Indonesian ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_fleurs_id_30", "display": "Fleurs-ID-30 [SEA]", "description": "Fleurs Indonesian ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_cv21_id_30": 0.04782021, "asr_fleurs_id_30": 0.06179733, "average": 0.0828, "commonvoice_17_id_asr": 0.05023912, "gigaspeech2_id_test": 0.17125355 }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2", "average": 0.0879, "commonvoice_17_id_asr": 0.05246785, "gigaspeech2_id_test": 0.166, "asr_cv21_id_30": 0.04934099, "asr_fleurs_id_30": 0.08390553 }, { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_cv21_id_30": 0.05947955, "asr_fleurs_id_30": 0.06564551, "average": 0.0899, "commonvoice_17_id_asr": 0.06156846, "gigaspeech2_id_test": 0.17304384 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 0.098, "commonvoice_17_id_asr": 0.079, "gigaspeech2_id_test": 0.163, "asr_cv21_id_30": 0.08246029, "asr_fleurs_id_30": 0.06760733 }, { "model": "MERaLiON-3-3B-ASR-CTM", "asr_cv21_id_30": 0.07688408, "asr_fleurs_id_30": 0.07424734, "commonvoice_17_id_asr": 0.07503366, "gigaspeech2_id_test": 0.1853126, "average": 0.1029 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "average": 0.103, "commonvoice_17_id_asr": 0.078, "gigaspeech2_id_test": 0.196, "asr_cv21_id_30": 0.0737, "asr_fleurs_id_30": 0.0644 }, { "model": "Fun-ASR-MLT-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512", "asr_cv21_id_30": 0.07840487, "asr_fleurs_id_30": 0.07779371, "average": 0.1092, "commonvoice_17_id_asr": 0.0816734, "gigaspeech2_id_test": 0.19902061 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 0.1102, "commonvoice_17_id_asr": 0.085, "gigaspeech2_id_test": 0.178, "asr_cv21_id_30": 0.08888138, "asr_fleurs_id_30": 0.08873463 }, { "model": "MERaLiON-3-10B", "asr_cv21_id_30": 0.07654613, "asr_fleurs_id_30": 0.07771825, "commonvoice_17_id_asr": 0.08399499, "average": 0.1127, "gigaspeech2_id_test": 0.21246534, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 0.1165, "commonvoice_17_id_asr": 0.113, "gigaspeech2_id_test": 0.172, "asr_cv21_id_30": 0.11676242, "asr_fleurs_id_30": 0.06413642 }, { "model": "Omnilingual-LLM-ASR-7B [with language code]", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.1338, "commonvoice_17_id_asr": 0.119, "gigaspeech2_id_test": 0.244, "asr_cv21_id_30": 0.11591754, "asr_fleurs_id_30": 0.05621369 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 0.1353, "commonvoice_17_id_asr": 0.11, "gigaspeech2_id_test": 0.227, "asr_cv21_id_30": 0.09783711, "asr_fleurs_id_30": 0.10646646 }, { "model": "MERaLiON-3-3B-ASR", "asr_cv21_id_30": 0.16880703, "asr_fleurs_id_30": 0.08337735, "commonvoice_17_id_asr": 0.14027023, "gigaspeech2_id_test": 0.18855969, "average": 0.1453, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_cv21_id_30": 0.12402839, "asr_fleurs_id_30": 0.0686637, "average": 0.1488, "commonvoice_17_id_asr": 0.1307053, "gigaspeech2_id_test": 0.27168533 }, { "model": "Omnilingual-LLM-ASR-7B", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.1541, "commonvoice_17_id_asr": 0.153, "gigaspeech2_id_test": 0.259, "asr_cv21_id_30": 0.14582629, "asr_fleurs_id_30": 0.05870369 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 0.156, "commonvoice_17_id_asr": 0.136, "gigaspeech2_id_test": 0.275, "asr_cv21_id_30": 0.11304495, "asr_fleurs_id_30": 0.10005282 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 0.1677, "commonvoice_17_id_asr": 0.124, "gigaspeech2_id_test": 0.316, "asr_cv21_id_30": 0.10476512, "asr_fleurs_id_30": 0.12600921 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 0.2796, "commonvoice_17_id_asr": 0.26, "gigaspeech2_id_test": 0.337, "asr_cv21_id_30": 0.28050017, "asr_fleurs_id_30": 0.24077567 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 0.4054, "asr_cv21_id_30": 0.33964177, "asr_fleurs_id_30": 0.34626122, "gigaspeech2_id_test": 0.58782954, "commonvoice_17_id_asr": 0.34786646 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_cv21_id_30": 0.52230483, "asr_fleurs_id_30": 0.11763374, "commonvoice_17_id_asr": 0.3899336, "gigaspeech2_id_test": 0.64373223, "average": 0.4184, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 0.451, "asr_cv21_id_30": 0.42227104, "asr_fleurs_id_30": 0.3482985, "gigaspeech2_id_test": 0.63825605, "commonvoice_17_id_asr": 0.39522682 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "asr_cv21_id_30": 0.54782021, "asr_fleurs_id_30": 0.07228552, "average": 0.4712, "commonvoice_17_id_asr": 0.58671124, "gigaspeech2_id_test": 0.67811633 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_cv21_id_30": 0.79553903, "asr_fleurs_id_30": 0.09416736, "average": 0.6049, "commonvoice_17_id_asr": 0.82309514, "gigaspeech2_id_test": 0.70670832 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "asr_cv21_id_30": 0.72220345, "asr_fleurs_id_30": 0.6942579, "average": 0.7466, "commonvoice_17_id_asr": 0.75284394, "gigaspeech2_id_test": 0.81726753 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 1.0618, "commonvoice_17_id_asr": 1.189, "gigaspeech2_id_test": 2.118, "asr_cv21_id_30": 0.51858736, "asr_fleurs_id_30": 0.42179129 }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_cv21_id_30": 1.15833052, "asr_fleurs_id_30": 1.23760658, "commonvoice_17_id_asr": 1.17077587, "gigaspeech2_id_test": 1.22375118, "average": 1.1976 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "commonvoice_17_id_asr": 1.15791429, "average": 1.2216, "gigaspeech2_id_test": 1.37889213, "asr_cv21_id_30": 1.12994255, "asr_fleurs_id_30": 1.21957293 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 2.3754, "commonvoice_17_id_asr": 1.327, "gigaspeech2_id_test": 5.804, "asr_cv21_id_30": 1.25684353, "asr_fleurs_id_30": 1.1137101 } ], "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "ASR-Thai", "title": "Task: Automatic Speech Recognition - Thai", "taskName": "asr_thai", "metric": "cer", "metricInfo": "Character Error Rate (CER) - the lower, the better. CER is used for every column in this table. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. SALMONN-7B's scores in this table reflect an output-script limitation, not transcription accuracy: its Vicuna-7B backbone answers in Chinese for Thai audio (72% of predictions were CJK on asr_cv21_th_30, despite prompts specifying Thai), so error rates exceed 1.0. The same model scores 0.06-0.25 on Latin-script cells. MERaLiON-SpeechEncoder2-ASR-CTC returns an empty transcript on 13.4% of clips in gigaspeech2_th_test, and those count as full deletions in the error rate. This is the model's own output, not a failed run: two independent evaluations produced byte-identical predictions, and the clips it answers are transcribed normally. Scored over only the clips it does answer, the error rate is roughly 3-5 points lower.", "ascending": true, "datasets": [ { "display": "GigaSpeech-2-Thai", "internal": "gigaspeech2_th_test", "description": "Thai subset of the GigaSpeech 2 corpus for ASR benchmarking, covering varied speakers and recording conditions to test recognition robustness.", "hfLink": "https://huggingface.co/datasets/speechcolab/gigaspeech2", "stats": { "num_rows": 5000, "audio_length": { "min": 0.11, "max": 14.02, "mean": 4.12, "median": 3.73, "std": 2.73, "total_hours": 5.72, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 5000, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Could you put the speech into writing in Thai?", "answer": "คุณฐิติพงศ์เกิดทองทวีนะครับคุณยศศิริใบศรีนะครับหรือคุณหัวกลมนะครับแล้วก็คุณวรรธนภูมิลายสุวรรณชัยครับ", "audioFile": "examples/gigaspeech2_th_test/example_0.wav" }, { "instruction": "Write down what was said in Thai, please.", "answer": "อุ๊ยเกลือเกลือแบบนั้นอะแบบไม่ค่อยมีแล้วนะไม่ค่อยมีขายแล้วนะเกลือซองสีชมพูอะ", "audioFile": "examples/gigaspeech2_th_test/example_1.wav" } ] }, { "display": "Lotus-Thai", "internal": "lotus_thai_th_30_asr", "description": "Thai speech dataset based on the LOTUS corpus, commonly used for Thai ASR evaluation with phonetically balanced speech material.", "hfLink": "https://ieeexplore.ieee.org/document/5278377", "stats": { "num_rows": 501, "audio_length": { "min": 1.84, "max": 29.78, "mean": 8.78, "median": 7.66, "std": 5.17, "total_hours": 1.22, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 501, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Please transcribe the following speech into Thai text.", "answer": "เพราะ ทั้ง พรรค ความ หวัง ใหม่ และ พรรค เอกภาพ มี เป้าหมาย ใน ทาง การเมือง ที่ ตรง กัน อยู่ ข้อ หนึ่ง คือ ต้องการ หา ทาง โค่น ล้ม รัฐบาล พลเอก ชาติชาย", "audioFile": "examples/lotus_thai_th_30_asr/example_0.wav" }, { "instruction": "Please transcribe the following speech into Thai text.", "answer": "เรา ไม่ ได้ กำลัง รบ กับ คน ชาติ อื่น ใด", "audioFile": "examples/lotus_thai_th_30_asr/example_1.wav" } ] }, { "display": "Cv21-TH-30 [SEA]", "internal": "asr_cv21_th_30", "description": "CV21 Thai ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Fleurs-TH-30 [SEA]", "internal": "asr_fleurs_th_30", "description": "Fleurs Thai ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "THAI-Elderly-TH-30 [SEA]", "internal": "asr_thai_elderly_th_30", "description": "Thai Elderly ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "THAI-Lotus-30 [SEA]", "internal": "asr_thai_lotus_30", "description": "Thai Lotus ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Commonvoice-17-TH-ASR", "internal": "commonvoice_17_th_asr", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "stats": {} } ], "data": { "columns": [ { "key": "gigaspeech2_th_test", "display": "GigaSpeech-2-Thai", "description": "Thai subset of the GigaSpeech 2 corpus for ASR benchmarking, covering varied speakers and recording conditions to test recognition robustness.", "hfLink": "https://huggingface.co/datasets/speechcolab/gigaspeech2", "isWer": true }, { "key": "lotus_thai_th_30_asr", "display": "Lotus-Thai", "description": "Thai speech dataset based on the LOTUS corpus, commonly used for Thai ASR evaluation with phonetically balanced speech material.", "hfLink": "https://ieeexplore.ieee.org/document/5278377", "isWer": true }, { "key": "asr_cv21_th_30", "display": "Cv21-TH-30 [SEA]", "description": "CV21 Thai ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_fleurs_th_30", "display": "Fleurs-TH-30 [SEA]", "description": "Fleurs Thai ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_thai_elderly_th_30", "display": "THAI-Elderly-TH-30 [SEA]", "description": "Thai Elderly ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_thai_lotus_30", "display": "THAI-Lotus-30 [SEA]", "description": "Thai Lotus ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "commonvoice_17_th_asr", "display": "Commonvoice-17-TH-ASR", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "isWer": true } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_cv21_th_30": 0.0291, "asr_fleurs_th_30": 0.05, "asr_thai_elderly_th_30": 0.0232, "asr_thai_lotus_30": 0.0141, "average": 0.0443, "gigaspeech2_th_test": 0.148, "commonvoice_17_th_asr": 0.0321, "lotus_thai_th_30_asr": 0.0137 }, { "model": "Fun-ASR-MLT-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512", "asr_cv21_th_30": 0.0085, "asr_fleurs_th_30": 0.0771, "asr_thai_elderly_th_30": 0.0046, "asr_thai_lotus_30": 0.0093, "average": 0.0446, "gigaspeech2_th_test": 0.1953, "lotus_thai_th_30_asr": 0.0096, "commonvoice_17_th_asr": 0.0081 }, { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_cv21_th_30": 0.0385, "asr_fleurs_th_30": 0.0863, "asr_thai_elderly_th_30": 0.0254, "asr_thai_lotus_30": 0.0166, "average": 0.0509, "commonvoice_17_th_asr": 0.036, "gigaspeech2_th_test": 0.137, "lotus_thai_th_30_asr": 0.0166 }, { "model": "Omnilingual-LLM-ASR-7B [with language code]", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.0653, "asr_cv21_th_30": 0.0479, "asr_fleurs_th_30": 0.0685, "asr_thai_elderly_th_30": 0.0388, "asr_thai_lotus_30": 0.0334, "gigaspeech2_th_test": 0.1886, "lotus_thai_th_30_asr": 0.0331, "commonvoice_17_th_asr": 0.0466 }, { "model": "MERaLiON-3-3B-ASR-CTM", "asr_cv21_th_30": 0.0805, "asr_fleurs_th_30": 0.0931, "asr_thai_elderly_th_30": 0.0631, "asr_thai_lotus_30": 0.0138, "commonvoice_17_th_asr": 0.0761, "gigaspeech2_th_test": 0.1513, "lotus_thai_th_30_asr": 0.0137, "average": 0.0702 }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2", "average": 0.0717, "lotus_thai_th_30_asr": 0.0224, "asr_cv21_th_30": 0.051, "asr_fleurs_th_30": 0.0913, "asr_thai_elderly_th_30": 0.0453, "asr_thai_lotus_30": 0.0237, "commonvoice_17_th_asr": 0.0494, "gigaspeech2_th_test": 0.2188 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 0.0761, "asr_cv21_th_30": 0.053, "asr_fleurs_th_30": 0.1047, "asr_thai_elderly_th_30": 0.0405, "asr_thai_lotus_30": 0.0286, "lotus_thai_th_30_asr": 0.0295, "gigaspeech2_th_test": 0.2208, "commonvoice_17_th_asr": 0.0555 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_cv21_th_30": 0.0804, "asr_fleurs_th_30": 0.065, "asr_thai_elderly_th_30": 0.057, "asr_thai_lotus_30": 0.0327, "average": 0.0801, "gigaspeech2_th_test": 0.2281, "lotus_thai_th_30_asr": 0.0256, "commonvoice_17_th_asr": 0.0722 }, { "model": "MERaLiON-3-10B", "asr_thai_lotus_30": 0.0119, "lotus_thai_th_30_asr": 0.0113, "average": 0.0876, "asr_cv21_th_30": 0.1024, "asr_thai_elderly_th_30": 0.062, "asr_fleurs_th_30": 0.1279, "commonvoice_17_th_asr": 0.1065, "gigaspeech2_th_test": 0.1914, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "average": 0.0885, "asr_cv21_th_30": 0.0778, "asr_fleurs_th_30": 0.0832, "asr_thai_elderly_th_30": 0.0556, "asr_thai_lotus_30": 0.0266, "commonvoice_17_th_asr": 0.0726, "gigaspeech2_th_test": 0.2772, "lotus_thai_th_30_asr": 0.0267 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 0.1224, "asr_cv21_th_30": 0.1483, "asr_fleurs_th_30": 0.1164, "asr_thai_elderly_th_30": 0.1661, "asr_thai_lotus_30": 0.0267, "lotus_thai_th_30_asr": 0.0276, "commonvoice_17_th_asr": 0.1217, "gigaspeech2_th_test": 0.2498 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 0.1288, "asr_cv21_th_30": 0.1441, "asr_fleurs_th_30": 0.1483, "asr_thai_elderly_th_30": 0.14, "asr_thai_lotus_30": 0.0413, "lotus_thai_th_30_asr": 0.0382, "commonvoice_17_th_asr": 0.1375, "gigaspeech2_th_test": 0.252 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 0.1551, "asr_cv21_th_30": 0.3645, "asr_fleurs_th_30": 0.1323, "asr_thai_elderly_th_30": 0.1893, "asr_thai_lotus_30": 0.0143, "lotus_thai_th_30_asr": 0.0275, "commonvoice_17_th_asr": 0.1065, "gigaspeech2_th_test": 0.2516 }, { "model": "MERaLiON-3-3B-ASR", "asr_cv21_th_30": 0.3081, "asr_fleurs_th_30": 0.0868, "asr_thai_elderly_th_30": 0.2472, "asr_thai_lotus_30": 0.0142, "commonvoice_17_th_asr": 0.2709, "gigaspeech2_th_test": 0.2027, "lotus_thai_th_30_asr": 0.0144, "average": 0.1635, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_cv21_th_30": 0.2685, "asr_fleurs_th_30": 0.1121, "asr_thai_elderly_th_30": 0.1423, "asr_thai_lotus_30": 0.0411, "commonvoice_17_th_asr": 0.2164, "gigaspeech2_th_test": 0.5644, "lotus_thai_th_30_asr": 0.0852, "average": 0.2043, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 0.1934, "asr_cv21_th_30": 0.4955, "asr_fleurs_th_30": 0.1348, "asr_thai_elderly_th_30": 0.2928, "asr_thai_lotus_30": 0.0205, "lotus_thai_th_30_asr": 0.0254, "gigaspeech2_th_test": 0.2201, "commonvoice_17_th_asr": 0.1646 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 0.2317, "asr_cv21_th_30": 0.6373, "asr_fleurs_th_30": 0.1299, "asr_thai_elderly_th_30": 0.4814, "asr_thai_lotus_30": 0.0117, "lotus_thai_th_30_asr": 0.0162, "commonvoice_17_th_asr": 0.1495, "gigaspeech2_th_test": 0.1962 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_cv21_th_30": 0.3725, "asr_fleurs_th_30": 0.2078, "asr_thai_elderly_th_30": 0.4464, "asr_thai_lotus_30": 0.2881, "average": 0.3257, "lotus_thai_th_30_asr": 0.1406, "commonvoice_17_th_asr": 0.3216, "gigaspeech2_th_test": 0.5029 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 0.3659, "asr_cv21_th_30": 0.3829, "asr_fleurs_th_30": 0.3594, "asr_thai_elderly_th_30": 0.3406, "asr_thai_lotus_30": 0.2881, "lotus_thai_th_30_asr": 0.2884, "commonvoice_17_th_asr": 0.3781, "gigaspeech2_th_test": 0.5238 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "asr_cv21_th_30": 1.0145, "asr_fleurs_th_30": 0.17, "asr_thai_elderly_th_30": 0.7269, "asr_thai_lotus_30": 0.3643, "average": 0.4937, "lotus_thai_th_30_asr": 0.2098, "commonvoice_17_th_asr": 0.2977, "gigaspeech2_th_test": 0.673 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 0.5213, "asr_cv21_th_30": 0.4282, "commonvoice_17_th_asr": 0.4349, "lotus_thai_th_30_asr": 0.4293, "asr_thai_elderly_th_30": 0.6942, "asr_thai_lotus_30": 0.4177, "gigaspeech2_th_test": 0.7832, "asr_fleurs_th_30": 0.4619 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 0.5718, "asr_cv21_th_30": 0.5049, "commonvoice_17_th_asr": 0.5452, "lotus_thai_th_30_asr": 0.3868, "asr_thai_elderly_th_30": 0.7299, "asr_thai_lotus_30": 0.4427, "gigaspeech2_th_test": 0.8374, "asr_fleurs_th_30": 0.5555 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "asr_cv21_th_30": 1.0444, "asr_fleurs_th_30": 1.0675, "asr_thai_elderly_th_30": 1.0521, "asr_thai_lotus_30": 1.0182, "average": 1.052, "lotus_thai_th_30_asr": 1.053, "gigaspeech2_th_test": 1.0904, "commonvoice_17_th_asr": 1.0386 }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_cv21_th_30": 1.0733, "asr_fleurs_th_30": 1.1242, "asr_thai_elderly_th_30": 1.0799, "asr_thai_lotus_30": 1.0613, "commonvoice_17_th_asr": 1.0775, "gigaspeech2_th_test": 1.0976, "lotus_thai_th_30_asr": 1.0343, "average": 1.0783 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 1.1021, "asr_cv21_th_30": 1.024, "asr_fleurs_th_30": 1.0392, "asr_thai_elderly_th_30": 1.092, "asr_thai_lotus_30": 1.153, "lotus_thai_th_30_asr": 1.1042, "commonvoice_17_th_asr": 1.0265, "gigaspeech2_th_test": 1.2759 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "average": 1.1725, "asr_cv21_th_30": 1.2789, "asr_fleurs_th_30": 1.022, "asr_thai_elderly_th_30": 1.3787, "asr_thai_lotus_30": 1.0669, "lotus_thai_th_30_asr": 1.0704, "commonvoice_17_th_asr": 1.2762, "gigaspeech2_th_test": 1.1142 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 2.1418, "asr_cv21_th_30": 2.4883, "asr_thai_elderly_th_30": 2.2595, "asr_thai_lotus_30": 3.1587, "lotus_thai_th_30_asr": 3.1061, "asr_fleurs_th_30": 1.4638, "commonvoice_17_th_asr": 1.2162, "gigaspeech2_th_test": 1.3001 } ], "metric": "cer", "metricInfo": "Character Error Rate (CER) - the lower, the better. CER is used for every column in this table. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. SALMONN-7B's scores in this table reflect an output-script limitation, not transcription accuracy: its Vicuna-7B backbone answers in Chinese for Thai audio (72% of predictions were CJK on asr_cv21_th_30, despite prompts specifying Thai), so error rates exceed 1.0. The same model scores 0.06-0.25 on Latin-script cells. MERaLiON-SpeechEncoder2-ASR-CTC returns an empty transcript on 13.4% of clips in gigaspeech2_th_test, and those count as full deletions in the error rate. This is the model's own output, not a failed run: two independent evaluations produced byte-identical predictions, and the clips it answers are transcribed normally. Scored over only the clips it does answer, the error rate is roughly 3-5 points lower.", "ascending": true } }, { "key": "ASR-Vietnamese", "title": "Task: Automatic Speech Recognition - Vietnamese", "taskName": "asr_vietnamese", "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. MERaLiON-SpeechEncoder2-ASR-CTC returns an empty transcript on 5.1-6.2% of clips in asr_bud500_30 and asr_vietmed_vi_30, and those count as full deletions in the error rate. This is the model's own output, not a failed run: two independent evaluations produced byte-identical predictions, and the clips it answers are transcribed normally. Scored over only the clips it does answer, the error rate is roughly 3-5 points lower.", "ascending": true, "datasets": [ { "display": "CommonVoice-17-Vietnamese", "internal": "commonvoice_17_vi_asr", "description": "Speech Recognition dataset", "hfLink": null, "stats": { "num_rows": 1274, "audio_length": { "min": 1.37, "max": 10.03, "mean": 3.8, "median": 3.67, "std": 1.1, "total_hours": 1.35, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 1274, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Please transcribe.", "answer": "Nên cô dịu giọng hỏi lại", "audioFile": "examples/commonvoice_17_vi_asr/example_0.wav" }, { "instruction": "Please transcribe.", "answer": "Em cố gắng nhắm mắt lại để ngủ tiếp", "audioFile": "examples/commonvoice_17_vi_asr/example_1.wav" } ] }, { "display": "GigaSpeech-2-Vietnamese", "internal": "gigaspeech2_vi_test", "description": "Vietnamese subset of the GigaSpeech 2 corpus used for ASR evaluation, with diverse speech content that reflects practical transcription scenarios.", "hfLink": "https://huggingface.co/datasets/speechcolab/gigaspeech2", "stats": { "num_rows": 5000, "audio_length": { "min": 5.0, "max": 15.0, "mean": 9.01, "median": 8.64, "std": 2.69, "total_hours": 12.51, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 5000, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Record the speech in written form in Vietnamese.", "answer": "THÔNG QUA HOẠT ĐỘNG TỔ CHỨC FAMTRIP TRUNG TÂM DU LỊCH PHONG NHA KẺ BÀNG HY VỌNG MANG ĐẾN CHO ĐỐI TÁC LÀ CÁC ĐƠN VỊ LỮ HÀNH NHỮNG TRẢI NGHIỆM THỰC TẾ VỀ CÁC SẢN PHẨM DU LỊCH ĐẶC SẮC MỚI CỦA PHONG NHA MIỀN DI SẢN DIỆU KỲ", "audioFile": "examples/gigaspeech2_vi_test/example_0.wav" }, { "instruction": "Could you put the speech into writing in Vietnamese?", "answer": "ĐI TRƯỚC THỜI GIAN ĐẾN BỐN MƯƠI NĂM VỀ PHƯƠNG DIỆN TÀI CHÍNH THƯƠNG MẠI KỸ THUẬT VÀ VĂN MINH", "audioFile": "examples/gigaspeech2_vi_test/example_1.wav" } ] }, { "display": "Bud500-30 [SEA]", "internal": "asr_bud500_30", "description": "BUD500 Vietnamese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Cv21-VI-30 [SEA]", "internal": "asr_cv21_vi_30", "description": "CV21 Vietnamese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Fleurs-VI-30 [SEA]", "internal": "asr_fleurs_vi_30", "description": "Fleurs Vietnamese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Vietmed-VI-30 [SEA]", "internal": "asr_vietmed_vi_30", "description": "VietMed Vietnamese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} } ], "data": { "columns": [ { "key": "commonvoice_17_vi_asr", "display": "CommonVoice-17-Vietnamese", "description": "Speech Recognition dataset", "hfLink": null, "isWer": true }, { "key": "gigaspeech2_vi_test", "display": "GigaSpeech-2-Vietnamese", "description": "Vietnamese subset of the GigaSpeech 2 corpus used for ASR evaluation, with diverse speech content that reflects practical transcription scenarios.", "hfLink": "https://huggingface.co/datasets/speechcolab/gigaspeech2", "isWer": true }, { "key": "asr_bud500_30", "display": "Bud500-30 [SEA]", "description": "BUD500 Vietnamese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_cv21_vi_30", "display": "Cv21-VI-30 [SEA]", "description": "CV21 Vietnamese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_fleurs_vi_30", "display": "Fleurs-VI-30 [SEA]", "description": "Fleurs Vietnamese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_vietmed_vi_30", "display": "Vietmed-VI-30 [SEA]", "description": "VietMed Vietnamese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_bud500_30": 0.04003096, "asr_cv21_vi_30": 0.39616263, "asr_fleurs_vi_30": 0.04589686, "asr_vietmed_vi_30": 0.18896909, "average": 0.1407, "commonvoice_17_vi_asr": 0.09950834, "gigaspeech2_vi_test": 0.07338368 }, { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_bud500_30": 0.04533894, "asr_cv21_vi_30": 0.40648698, "asr_fleurs_vi_30": 0.07591695, "asr_vietmed_vi_30": 0.19780677, "average": 0.1522, "commonvoice_17_vi_asr": 0.1061351, "gigaspeech2_vi_test": 0.08148708 }, { "model": "Fun-ASR-MLT-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512", "asr_bud500_30": 0.05529139, "asr_cv21_vi_30": 0.41123801, "asr_fleurs_vi_30": 0.08064453, "asr_vietmed_vi_30": 0.20669073, "average": 0.1633, "commonvoice_17_vi_asr": 0.11660966, "gigaspeech2_vi_test": 0.10951496 }, { "model": "MERaLiON-3-3B-ASR-CTM", "asr_bud500_30": 0.04655535, "asr_cv21_vi_30": 0.43855642, "asr_fleurs_vi_30": 0.09715164, "asr_vietmed_vi_30": 0.2043772, "commonvoice_17_vi_asr": 0.14739205, "gigaspeech2_vi_test": 0.07673541, "average": 0.1685 }, { "model": "MERaLiON-3-3B-ASR", "asr_bud500_30": 0.04832467, "asr_cv21_vi_30": 0.43417085, "asr_fleurs_vi_30": 0.10333688, "asr_vietmed_vi_30": 0.20396076, "commonvoice_17_vi_asr": 0.14354425, "gigaspeech2_vi_test": 0.0811095, "average": 0.1691, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 0.1834, "commonvoice_17_vi_asr": 0.17058572, "gigaspeech2_vi_test": 0.095, "asr_bud500_30": 0.06037819, "asr_cv21_vi_30": 0.44723618, "asr_fleurs_vi_30": 0.09888508, "asr_vietmed_vi_30": 0.22839163 }, { "model": "MERaLiON-3-10B", "asr_bud500_30": 0.05153157, "asr_fleurs_vi_30": 0.09723043, "average": 0.2028, "asr_cv21_vi_30": 0.4648698, "asr_vietmed_vi_30": 0.25268369, "commonvoice_17_vi_asr": 0.20350577, "gigaspeech2_vi_test": 0.14676155, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "Omnilingual-LLM-ASR-7B [with language code]", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.2048, "commonvoice_17_vi_asr": 0.126, "gigaspeech2_vi_test": 0.147, "asr_bud500_30": 0.18810129, "asr_cv21_vi_30": 0.41955231, "asr_fleurs_vi_30": 0.09742741, "asr_vietmed_vi_30": 0.25097168 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 0.2116, "commonvoice_17_vi_asr": 0.18266353, "gigaspeech2_vi_test": 0.113, "asr_bud500_30": 0.09863983, "asr_cv21_vi_30": 0.46130653, "asr_fleurs_vi_30": 0.11830753, "asr_vietmed_vi_30": 0.2956228 }, { "model": "Omnilingual-LLM-ASR-7B", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.2172, "commonvoice_17_vi_asr": 0.142, "gigaspeech2_vi_test": 0.152, "asr_bud500_30": 0.22359836, "asr_cv21_vi_30": 0.43033349, "asr_fleurs_vi_30": 0.09801836, "asr_vietmed_vi_30": 0.25754211 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_bud500_30": 0.20634745, "asr_cv21_vi_30": 0.48761992, "asr_fleurs_vi_30": 0.10778868, "asr_vietmed_vi_30": 0.25666297, "average": 0.2407, "commonvoice_17_vi_asr": 0.23375374, "gigaspeech2_vi_test": 0.15197793 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 0.2421, "commonvoice_17_vi_asr": 0.23674647, "gigaspeech2_vi_test": 0.189, "asr_bud500_30": 0.10770762, "asr_cv21_vi_30": 0.48369118, "asr_fleurs_vi_30": 0.16778947, "asr_vietmed_vi_30": 0.26790672 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 0.2427, "commonvoice_17_vi_asr": 0.24016674, "gigaspeech2_vi_test": 0.177, "asr_bud500_30": 0.07939843, "asr_cv21_vi_30": 0.48944724, "asr_fleurs_vi_30": 0.12051373, "asr_vietmed_vi_30": 0.34948177 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 0.2525, "commonvoice_17_vi_asr": 0.37654981, "gigaspeech2_vi_test": 0.227, "asr_bud500_30": 0.07585978, "asr_cv21_vi_30": 0.47693011, "asr_fleurs_vi_30": 0.08812985, "asr_vietmed_vi_30": 0.27082177 }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2", "average": 0.2555, "commonvoice_17_vi_asr": 0.21601112, "gigaspeech2_vi_test": 0.132, "asr_cv21_vi_30": 0.48698036, "asr_fleurs_vi_30": 0.14387582, "asr_bud500_30": 0.23576247, "asr_vietmed_vi_30": 0.31824912 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 0.2622, "commonvoice_17_vi_asr": 0.17315092, "gigaspeech2_vi_test": 0.168, "asr_bud500_30": 0.19517859, "asr_cv21_vi_30": 0.45034262, "asr_fleurs_vi_30": 0.13398731, "asr_vietmed_vi_30": 0.45238756 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "average": 0.3936, "commonvoice_17_vi_asr": 0.137, "gigaspeech2_vi_test": 0.16, "asr_bud500_30": 0.7942, "asr_cv21_vi_30": 0.4608, "asr_fleurs_vi_30": 0.0825, "asr_vietmed_vi_30": 0.727 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_bud500_30": 0.69412805, "asr_cv21_vi_30": 0.78775697, "asr_fleurs_vi_30": 0.17369893, "asr_vietmed_vi_30": 0.36706459, "commonvoice_17_vi_asr": 0.67977768, "gigaspeech2_vi_test": 0.20035434, "average": 0.4838, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 0.5809, "asr_bud500_30": 0.66017914, "asr_cv21_vi_30": 0.7311101, "asr_vietmed_vi_30": 0.66069776, "commonvoice_17_vi_asr": 0.60592133, "gigaspeech2_vi_test": 0.44261981, "asr_fleurs_vi_30": 0.38498207 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 0.6421, "commonvoice_17_vi_asr": 0.67037195, "gigaspeech2_vi_test": 0.982, "asr_bud500_30": 0.5307973, "asr_cv21_vi_30": 0.78382823, "asr_fleurs_vi_30": 0.35066777, "asr_vietmed_vi_30": 0.5349343 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 0.8132, "asr_bud500_30": 1.55202919, "asr_cv21_vi_30": 0.76774783, "asr_fleurs_vi_30": 0.38762164, "asr_vietmed_vi_30": 1.05344253, "commonvoice_17_vi_asr": 0.68373236, "gigaspeech2_vi_test": 0.43469649 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "asr_cv21_vi_30": 1.11009593, "asr_fleurs_vi_30": 0.15612812, "average": 0.9088, "commonvoice_17_vi_asr": 1.15166738, "asr_bud500_30": 2.23708946, "asr_vietmed_vi_30": 0.58467518, "gigaspeech2_vi_test": 0.21339529 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "asr_bud500_30": 1.0629216, "asr_cv21_vi_30": 1.003746, "asr_fleurs_vi_30": 1.05909467, "asr_vietmed_vi_30": 1.00305386, "average": 1.0347, "commonvoice_17_vi_asr": 1.00470286, "gigaspeech2_vi_test": 1.07442347 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 1.0537, "commonvoice_17_vi_asr": 1.496, "gigaspeech2_vi_test": 1.546, "asr_bud500_30": 0.83158244, "asr_cv21_vi_30": 0.94755596, "asr_vietmed_vi_30": 0.84365167, "asr_fleurs_vi_30": 0.65713273 }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_bud500_30": 1.07420104, "asr_cv21_vi_30": 1.03535861, "asr_fleurs_vi_30": 1.18413899, "asr_vietmed_vi_30": 1.17476402, "commonvoice_17_vi_asr": 1.04884566, "gigaspeech2_vi_test": 1.20174848, "average": 1.1198 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "asr_bud500_30": 1.41413248, "asr_cv21_vi_30": 1.28588397, "asr_fleurs_vi_30": 1.07118938, "asr_vietmed_vi_30": 1.15412734, "average": 1.235, "commonvoice_17_vi_asr": 1.42304404, "gigaspeech2_vi_test": 1.06135347 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_cv21_vi_30": 1.21973504, "asr_fleurs_vi_30": 0.31722019, "average": 1.4787, "commonvoice_17_vi_asr": 1.47862334, "asr_bud500_30": 4.68970474, "asr_vietmed_vi_30": 0.83523043, "gigaspeech2_vi_test": 0.33185594 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 2.1001, "commonvoice_17_vi_asr": 1.71130825, "gigaspeech2_vi_test": 2.504, "asr_bud500_30": 1.56109698, "asr_cv21_vi_30": 1.30936501, "asr_fleurs_vi_30": 1.12811724, "asr_vietmed_vi_30": 4.38686841 } ], "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures. MERaLiON-SpeechEncoder2-ASR-CTC returns an empty transcript on 5.1-6.2% of clips in asr_bud500_30 and asr_vietmed_vi_30, and those count as full deletions in the error rate. This is the model's own output, not a failed run: two independent evaluations produced byte-identical predictions, and the clips it answers are transcribed normally. Scored over only the clips it does answer, the error rate is roughly 3-5 points lower.", "ascending": true } }, { "key": "ASR-Other", "title": "Task: Automatic Speech Recognition - Other Datasets", "taskName": "asr_private", "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "CNA", "internal": "cna_test", "description": "Speech Recognition dataset", "hfLink": null, "stats": { "num_rows": 2400, "audio_length": { "min": 0.54, "max": 8.18, "mean": 2.67, "median": 2.61, "std": 0.92, "total_hours": 1.78, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 2400, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Capture the speech in written format, please.", "answer": "OPENED ON BOXING DAY", "audioFile": "examples/cna_test/example_0.wav" }, { "instruction": "Please transcribe the speech.", "answer": "BEYOND IT'S TERRITORIAL WATERS ON A POTENTIAL COMBAT MISSION", "audioFile": "examples/cna_test/example_1.wav" } ] }, { "display": "IDPC", "internal": "idpc_test", "description": "Speech Recognition dataset", "hfLink": null, "stats": { "num_rows": 4, "audio_length": { "min": 524.48, "max": 592.8, "mean": 560.3, "median": 561.96, "std": 31.4, "total_hours": 0.62, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 0, 0, 0, 4, 0 ] } } } }, { "display": "Parliament", "internal": "parliament_test", "description": "Speech Recognition dataset", "hfLink": null, "stats": { "num_rows": 4, "audio_length": { "min": 168.04, "max": 10931.95, "mean": 4795.19, "median": 4040.38, "std": 4620.4, "total_hours": 5.33, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 0, 0, 1, 1, 2 ] } } } }, { "display": "UKUS-News", "internal": "ukusnews_test", "description": "Speech Recognition dataset", "hfLink": null, "stats": { "num_rows": 15, "audio_length": { "min": 87.77, "max": 1853.0, "mean": 602.62, "median": 504.52, "std": 440.71, "total_hours": 2.51, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 0, 2, 1, 11, 1 ] } } } }, { "display": "Mediacorp", "internal": "mediacorp_test", "description": "Speech Recognition dataset", "hfLink": null, "stats": { "num_rows": 84, "audio_length": { "min": 11.04, "max": 146.56, "mean": 49.61, "median": 40.74, "std": 27.12, "total_hours": 1.16, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 15, 67, 2, 0, 0 ] } } }, "examples": [ { "instruction": "Write down what was said, please.", "answer": "ONE COULD SAY THAT WITHIN THE CONTEXT OF PARLIAMENT THOUGH WHERE THERE IS THE RULING PARTY AND THERE ARE OPPOSITION PARTIES THAT IS PAR FOR THE COURSE TO PROVE THAT YOUR POLICIES ARE THE BEST TO DEFEND YOUR POLICIES AND IN THAT PROCESS SHOULD WE REALLY EXPECT CHARITY AND AND REMARKS LIKE THAT'S A GREAT QUESTION YOU KNOW THERE THERE HAS TO BE I THINK SOME DEGREE OF BALANCE BECAUSE OTHERWISE IT'S GOING TO BACKFIRE YOU KNOW LIKE FOR EXAMPLE THE WHOLE SYLVIA LIM EPISODE I THINK THE CONSENSUS IS THAT IT BACKFIRED YOU SEE IT MADE YOU WON SILVIA A LOT OF SYMPATHY POINTS SEE THE FACT THAT YOU KNOW PEOPLE ARE TRYING TO MAKE HER APOLOGIZE FOR ASKING AN HONEST QUESTION YOU CAN SAY THAT HER QUESTION IS A STUPID QUESTION YOU CAN YOU CAN TRY AND CRITICIZE HER FOR THAT BUT TRYING TO MAKE HER APOLOGIZE FOR ASKING A QUESTION YOU KNOW SEEMS TO BE A LITTLE BIT LIKE SHOOTING YOURSELF IN THE FOOT SEE BECAUSE IT MAKES YOU COME ACROSS LIKE A BULLY OFTEN THE PAP MINISTERS AND MPS WOULD SAY IF YOU RAISE AN ISSUE THAT SEEMS TO CAST ASPERSIONS ON OUR INTENTIONS FOR THE CITIZENS OF SINGAPORE WE HAVE TO DEFEND IT WE HAVE TO TO MAKE SURE THAT THESE IDEAS ARE NOT ALLOWED TO PROPAGATE WITHIN SOCIETY SO THAT'S THEIR WAY OF DOING IT HOW WOULD YOU HAVE HANDLED IT YOU KNOW I'M A MEDICAL DOCTOR AND EVEN THOUGH I'M AN INFECTIOUS DISEASE DOCTOR I DO GENERAL MEDICINE SOMETIMES AND IN GENERAL MEDICINE SOMETIMES WE DEAL WITH INDIVIDUALS WHO HAVE PARANOIA YOU KNOW AND THESE PEOPLE YOU ASK THEM A SIMPLE QUESTION AND THEN THEY THINK THEY'RE OUT THERE TO TAKE AWAY THEIR LIFE AND TO DESTROY THEIR HOME AND THEIR FAMILY YOU KNOW AND AGAIN THAT'S AN EXTREME EXAMPLE BUT THAT SEEMS TO BE THE PAP'S APPROACH TO PEOPLE ASKING INNOCENT AND RELATIVELY STRAIGHTFORWARD QUESTIONS IN A PARLIAMENTARY DEMOCRACY IF SOMEBODY ASKS YOU A QUESTION YOU KNOW YOU CAN MAKE FUN OF THEM YOU CAN ANSWER THE QUESTION YOU CAN MAKE THEM LOOK REALLY SMALL OR REALLY SILLY BUT TO TREAT EVERY QUESTION AS AN ATTACK ON YOUR OWN PERSONAL INTEGRITY I MEAN TO ME THAT'S BORDERING A LITTLE BIT ON THE PARANOID", "audioFile": "examples/mediacorp_test/example_0.wav" }, { "instruction": "Could you write out what was said?", "answer": "WE LOOK OUT FOR SECURITY-RELATED ITEMS CONTROLLED AND PROHIBITED ITEMS NARCOTICS AS WELL AS EXPLOSIVES HOW DO YOU KNOW WHAT IS WHAT ORANGE ARE BASICALLY ORGANIC ITEMS FOR GREEN THIS COULD BE LIKE PLASTICS AND THEN FOR BLUE IS ACTUALLY METALLIC THINGS EXAMPLE THESE ARE DEODORANTS SO YOU WILL NOTICE THAT THE COLORS ARE OF THIS KIND SO IF LET'S SAY ONE DAY YOU WERE TO COME ACROSS OF A DIFFERENT COLOR THEN YOU CAN BE SUSPICIOUS SO IF THE IMAGE IS DEEMED TO BE SUSPICIOUS WE HAVE TO TAKE A CLOSER LOOK AND CONDUCT A PHYSICAL EXAMINATION WE NEED THE SINGPOST STAFF TO OPEN UP THE PACKAGES BY LAW ONLY POSTAL EMPLOYEES HAVE THE AUTHORITY TO OPEN MAIL ITEMS I ALWAYS KNEW THEY WERE POWERFUL AND A COMMON CONTRABAND IS THAT WE COME ACROSS WEAPONS LIKE THE KNUCKLE DUSTER KNIFE SWORDS TOBACCO PRODUCTS AND NARCOTICS SUCH AS CANNABIS", "audioFile": "examples/mediacorp_test/example_1.wav" } ] }, { "display": "IDPC-Short", "internal": "idpc_short_test", "description": "Speech Recognition dataset", "hfLink": null, "stats": { "num_rows": 122, "audio_length": { "min": 5.08, "max": 29.55, "mean": 15.48, "median": 14.5, "std": 6.88, "total_hours": 0.52, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 122, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Could you put the speech into writing?", "answer": ": It was a teaching session, : right, yah it's just : one way kind of session, so : and we didn't have practice before that. Yeah, we didn't practise before that also [lah]. We we were not very familiar with tools and all that, : okay : so there was a lot of downtime and lull periods where a lot of people were off camera trying to figure out what to do next, and and (uh) so there was a lot of quiet ses~ moments moments [ah] quiet moments : [mmhmm] : and (uh) not much interactions between the members, (uh) which kind of defeats the purpose [lah] of of the the whole session.", "audioFile": "examples/idpc_short_test/example_0.wav" }, { "instruction": "Capture the spoken content in writing.", "answer": ": Basically, do I recognize and sound out? (uh) yah Chiou-Har over to you : [mm] right I I think (uh) I will want to have (uh) an assessment of the person's teamwork because maybe I only (uh) do what my boss asked me to do.", "audioFile": "examples/idpc_short_test/example_1.wav" } ] }, { "display": "Mediacorp-Short", "internal": "mediacorp_short_test", "description": "Speech Recognition dataset", "hfLink": null, "stats": { "num_rows": 207, "audio_length": { "min": 5.02, "max": 29.92, "mean": 13.65, "median": 11.96, "std": 6.62, "total_hours": 0.78, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 207, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Record the spoken words in text form.", "answer": " Can you explain what are you doing? It is always polite. It is always gentle.", "audioFile": "examples/mediacorp_short_test/example_0.wav" }, { "instruction": "Capture the spoken content in writing.", "answer": " The thing I think I want to say is that before any conversation can actually take place, we have to kind of acknowledge the fact that we all have different starting points or understandings of it before these accusations start flying around.", "audioFile": "examples/mediacorp_short_test/example_1.wav" } ] }, { "display": "YouTube ASR: English Singapore Content", "internal": "ytb_asr_batch1", "description": "YouTube Evaluation Dataset for ASR Task: This dataset contains English and Singlish audio clips, featuring Singapore-related content. It includes approximately 2.5 hours of audio, with individual clips ranging from 2 seconds to 30 seconds in length.", "hfLink": null, "stats": { "num_rows": 384, "audio_length": { "min": 2.0, "max": 29.98, "mean": 23.4, "median": 24.72, "std": 4.71, "total_hours": 2.5, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 384, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Can you please transcribe the audio into text?", "answer": ": ..most fresh graduates, I was brimming with confidence, armed to the teeth with all the theories I had learned in school. (music) Then the Asian Financial Crisis struck, and all these theories turned to dust. : that sent shockwaves through Asian currency markets which saw : Singapore economy showed another round of none too good figures. : The Thai Baht and", "audioFile": "examples/ytb_asr_batch1/example_0.wav" }, { "instruction": "What is the text spoken in this audio?", "answer": ": I am taking the streets to find out if Singaporeans are comfortable with sex education being taught in schools, their experience with it, and whether they think anything should be changed. (music) First question", "audioFile": "examples/ytb_asr_batch1/example_1.wav" } ] }, { "display": "YouTube ASR: English with Strong Emotion", "internal": "ytb_asr_batch2", "description": "YouTube Evaluation Dataset for ASR Task: This dataset contains English, Singlish and some unknown languages audio clips, featuring speech with strong emotional expression. It includes approximately 3.9 hours of audio, with each clip lasting 30 seconds.", "hfLink": null, "stats": { "num_rows": 473, "audio_length": { "min": 30.0, "max": 30.0, "mean": 30.0, "median": 30.0, "std": 0.0, "total_hours": 3.94, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 0, 473, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Transcribe the input audio file into text.", "answer": ": we'll see how it goes. We are live now, hello everyone! Just want to set the stage a little bit So, Tesla is now the top selling car brand across most of europe, so today I've got my guest, Tom with us and our topic today is Tesla dominating Europe. Tom, (uh) Sochun is the president of the Tesla owners, West Sweden who's going to share his views about Tesla's growth in Europe, and also his recent experience visiting giga Berlin", "audioFile": "examples/ytb_asr_batch2/example_0.wav" }, { "instruction": "Transcribe the input audio file into text.", "answer": ": But minutes after broadcast, the news hotlines were ringing non-stop. First to call was Nancy Goh, a concerned mother. : Yes (uh), hello, my name is Nancy Goh and I'm a concerned mother. You know, as a mother of three, I am very disgusted that such a scene could be shown on TV at prime time. Prime time, you know, (uh) I I'm not finished yet (ah). You know, just like the other time I was watching this Can't Forget the Lyrics or Don't Forget the Lyrics or something like that, it was a celeb", "audioFile": "examples/ytb_asr_batch2/example_1.wav" } ] }, { "display": "Parliament-Short-TEST", "internal": "parliament_short_test", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "stats": {} }, { "display": "Ukusnews-Short-TEST", "internal": "ukusnews_short_test", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "stats": {} } ], "data": { "columns": [ { "key": "cna_test", "display": "CNA", "description": "Speech Recognition dataset", "hfLink": null, "isWer": true }, { "key": "idpc_short_test", "display": "IDPC-Short", "description": "Speech Recognition dataset", "hfLink": null, "isWer": true }, { "key": "idpc_test", "display": "IDPC", "description": "Speech Recognition dataset", "hfLink": null, "isWer": true }, { "key": "mediacorp_short_test", "display": "Mediacorp-Short", "description": "Speech Recognition dataset", "hfLink": null, "isWer": true }, { "key": "mediacorp_test", "display": "Mediacorp", "description": "Speech Recognition dataset", "hfLink": null, "isWer": true }, { "key": "parliament_test", "display": "Parliament", "description": "Speech Recognition dataset", "hfLink": null, "isWer": true }, { "key": "ukusnews_test", "display": "UKUS-News", "description": "Speech Recognition dataset", "hfLink": null, "isWer": true }, { "key": "ytb_asr_batch1", "display": "YouTube ASR: English Singapore Content", "description": "YouTube Evaluation Dataset for ASR Task: This dataset contains English and Singlish audio clips, featuring Singapore-related content. It includes approximately 2.5 hours of audio, with individual clips ranging from 2 seconds to 30 seconds in length.", "hfLink": null, "isWer": true }, { "key": "ytb_asr_batch2", "display": "YouTube ASR: English with Strong Emotion", "description": "YouTube Evaluation Dataset for ASR Task: This dataset contains English, Singlish and some unknown languages audio clips, featuring speech with strong emotional expression. It includes approximately 3.9 hours of audio, with each clip lasting 30 seconds.", "hfLink": null, "isWer": true }, { "key": "parliament_short_test", "display": "Parliament-Short-TEST", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "isWer": true }, { "key": "ukusnews_short_test", "display": "Ukusnews-Short-TEST", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "isWer": true } ], "rows": [ { "model": "Qwen3-ASR-1.7B", "parliament_test": 0.0505, "idpc_test": 0.1723, "ytb_asr_batch1": 0.0833, "idpc_short_test": 0.1493, "mediacorp_short_test": 0.1085, "ytb_asr_batch2": 0.0963, "ukusnews_test": 0.0486, "mediacorp_test": 0.0978, "cna_test": 0.1166, "average": 0.0929, "parliament_short_test": 0.048, "ukusnews_short_test": 0.0512, "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B" }, { "model": "MERaLiON-3-3B-ASR-CTM", "ytb_asr_batch1": 0.0776, "parliament_test": 0.05, "ytb_asr_batch2": 0.0918, "idpc_short_test": 0.1361, "mediacorp_test": 0.0969, "idpc_test": 0.1663, "ukusnews_test": 0.0582, "mediacorp_short_test": 0.1103, "cna_test": 0.132, "average": 0.0936, "parliament_short_test": 0.0471, "ukusnews_short_test": 0.0631 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 0.0981, "cna_test": 0.127, "idpc_short_test": 0.14, "idpc_test": 0.166, "mediacorp_short_test": 0.118, "mediacorp_test": 0.104, "parliament_test": 0.053, "ukusnews_test": 0.056, "ytb_asr_batch1": 0.092, "ytb_asr_batch2": 0.099, "ukusnews_short_test": 0.0704, "parliament_short_test": 0.0534 }, { "model": "MERaLiON-3-3B-ASR", "cna_test": 0.1894, "parliament_test": 0.0531, "mediacorp_short_test": 0.1115, "ytb_asr_batch2": 0.0948, "ukusnews_test": 0.0555, "ytb_asr_batch1": 0.0851, "idpc_test": 0.1579, "mediacorp_test": 0.1022, "idpc_short_test": 0.1434, "average": 0.1025, "parliament_short_test": 0.0578, "ukusnews_short_test": 0.0763, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 0.1031, "cna_test": 0.133, "idpc_short_test": 0.157, "idpc_test": 0.16, "mediacorp_short_test": 0.117, "mediacorp_test": 0.105, "parliament_test": 0.06, "ukusnews_test": 0.07, "ytb_asr_batch1": 0.098, "ytb_asr_batch2": 0.111, "parliament_short_test": 0.0511, "ukusnews_short_test": 0.0721 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "average": 0.106, "cna_test": 0.142, "idpc_short_test": 0.188, "idpc_test": 0.165, "mediacorp_short_test": 0.116, "mediacorp_test": 0.113, "parliament_test": 0.067, "ukusnews_test": 0.07, "ytb_asr_batch1": 0.107, "ytb_asr_batch2": 0.082, "parliament_short_test": 0.0569, "ukusnews_short_test": 0.0586 }, { "model": "Fun-ASR-Nano-2512", "cna_test": 0.1572, "average": 0.1104, "parliament_short_test": 0.0594, "ukusnews_short_test": 0.0627, "ukusnews_test": 0.058, "idpc_test": 0.1627, "parliament_test": 0.061, "ytb_asr_batch1": 0.0948, "mediacorp_short_test": 0.1289, "idpc_short_test": 0.1857, "ytb_asr_batch2": 0.1274, "mediacorp_test": 0.1168, "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512" }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2", "average": 0.1166, "cna_test": 0.163, "idpc_short_test": 0.175, "idpc_test": 0.186, "mediacorp_short_test": 0.118, "mediacorp_test": 0.105, "parliament_test": 0.062, "ukusnews_test": 0.08, "ytb_asr_batch1": 0.115, "ytb_asr_batch2": 0.119, "parliament_short_test": 0.062, "ukusnews_short_test": 0.0972 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 0.1216, "cna_test": 0.145, "idpc_short_test": 0.165, "idpc_test": 0.204, "mediacorp_short_test": 0.128, "mediacorp_test": 0.123, "parliament_test": 0.059, "ukusnews_test": 0.113, "ytb_asr_batch1": 0.107, "ytb_asr_batch2": 0.133, "parliament_short_test": 0.0582, "ukusnews_short_test": 0.1024 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "cna_test": 0.1174, "average": 0.1344, "ytb_asr_batch1": 0.0775, "parliament_short_test": 0.0469, "ukusnews_test": 0.2035, "ukusnews_short_test": 0.054, "parliament_test": 0.0785, "mediacorp_short_test": 0.1027, "idpc_short_test": 0.1638, "mediacorp_test": 0.1173, "idpc_test": 0.4297, "ytb_asr_batch2": 0.087, "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct" }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 0.1344, "cna_test": 0.135, "idpc_short_test": 0.151, "idpc_test": 0.177, "mediacorp_short_test": 0.121, "mediacorp_test": 0.123, "parliament_test": 0.185, "ukusnews_test": 0.174, "ytb_asr_batch1": 0.099, "ytb_asr_batch2": 0.16, "ukusnews_short_test": 0.0846, "parliament_short_test": 0.0692 }, { "model": "Fun-ASR-MLT-Nano-2512", "ytb_asr_batch1": 0.1614, "cna_test": 0.1462, "average": 0.1443, "parliament_short_test": 0.077, "ukusnews_short_test": 0.0971, "mediacorp_test": 0.1689, "ytb_asr_batch2": 0.1792, "mediacorp_short_test": 0.1679, "idpc_short_test": 0.1944, "idpc_test": 0.1959, "ukusnews_test": 0.1241, "parliament_test": 0.0753, "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512" }, { "model": "MERaLiON-3-10B", "ukusnews_test": 0.0817, "parliament_test": 0.0798, "mediacorp_test": 0.1273, "cna_test": 0.146, "idpc_test": 0.5311, "average": 0.1453, "ukusnews_short_test": 0.0705, "idpc_short_test": 0.1518, "parliament_short_test": 0.0561, "ytb_asr_batch2": 0.1479, "mediacorp_short_test": 0.112, "ytb_asr_batch1": 0.0942, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 0.1459, "cna_test": 0.174, "idpc_short_test": 0.211, "idpc_test": 0.199, "mediacorp_short_test": 0.148, "mediacorp_test": 0.152, "parliament_test": 0.1, "ukusnews_test": 0.091, "ytb_asr_batch1": 0.162, "ytb_asr_batch2": 0.245, "parliament_short_test": 0.0581, "ukusnews_short_test": 0.0643 }, { "model": "gemma-4-E4B-it", "ytb_asr_batch1": 0.1683, "ytb_asr_batch2": 0.1734, "cna_test": 0.4555, "average": 0.1874, "parliament_short_test": 0.085, "ukusnews_short_test": 0.089, "idpc_short_test": 0.2727, "mediacorp_test": 0.177, "mediacorp_short_test": 0.2119, "parliament_test": 0.1167, "idpc_test": 0.2353, "ukusnews_test": 0.077, "modelLink": "https://huggingface.co/google/gemma-4-E4B-it" }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 0.2078, "cna_test": 0.183, "idpc_short_test": 0.414, "idpc_test": 0.22, "mediacorp_short_test": 0.141, "mediacorp_test": 0.235, "parliament_test": 0.11, "ukusnews_test": 0.176, "ytb_asr_batch1": 0.174, "ytb_asr_batch2": 0.351, "parliament_short_test": 0.0966, "ukusnews_short_test": 0.1848 }, { "model": "Qwen2-Audio-7B-Instruct", "mediacorp_test": 0.2453, "ytb_asr_batch1": 0.1974, "parliament_short_test": 0.1063, "mediacorp_short_test": 0.2092, "idpc_short_test": 0.2779, "ukusnews_short_test": 0.1279, "ytb_asr_batch2": 0.2305, "average": 0.2175, "cna_test": 0.1961, "idpc_test": 0.3561, "ukusnews_test": 0.2361, "parliament_test": 0.2093, "modelLink": "https://arxiv.org/abs/2407.10759" }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 0.2225, "cna_test": 0.191, "idpc_short_test": 0.539, "idpc_test": 0.261, "mediacorp_short_test": 0.122, "mediacorp_test": 0.198, "parliament_test": 0.278, "ukusnews_test": 0.075, "ytb_asr_batch1": 0.169, "ytb_asr_batch2": 0.232, "parliament_short_test": 0.2365, "ukusnews_short_test": 0.1455 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 0.2396, "cna_test": 0.149, "idpc_short_test": 0.24, "idpc_test": 0.541, "mediacorp_short_test": 0.199, "mediacorp_test": 0.364, "parliament_test": 0.204, "ukusnews_test": 0.192, "ytb_asr_batch1": 0.221, "ytb_asr_batch2": 0.35, "parliament_short_test": 0.0884, "ukusnews_short_test": 0.0867 }, { "model": "canary-qwen-2.5b", "mediacorp_test": 0.3759, "idpc_test": 0.9634, "ytb_asr_batch1": 0.0895, "idpc_short_test": 0.1702, "mediacorp_short_test": 0.1135, "cna_test": 0.1366, "ytb_asr_batch2": 0.1385, "ukusnews_test": 0.9656, "average": 0.2829, "parliament_short_test": 0.0484, "ukusnews_short_test": 0.0564, "parliament_test": 0.0536, "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b" }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "ukusnews_short_test": 0.1662, "mediacorp_test": 0.2327, "idpc_short_test": 0.3515, "ukusnews_test": 0.1014, "parliament_short_test": 0.1247, "idpc_test": 0.3162, "ytb_asr_batch1": 0.2109, "parliament_test": 0.1328, "ytb_asr_batch2": 0.2158, "cna_test": 1.0423, "mediacorp_short_test": 0.2896, "average": 0.2895, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "Omnilingual-LLM-ASR-7B [with language code]", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.3237, "cna_test": 0.274, "idpc_short_test": 0.318, "idpc_test": 0.278, "mediacorp_short_test": 0.237, "mediacorp_test": 0.203, "parliament_test": 0.172, "ukusnews_test": 0.214, "ytb_asr_batch1": 0.84, "ytb_asr_batch2": 0.784, "parliament_short_test": 0.1109, "ukusnews_short_test": 0.1296 }, { "model": "Voxtral-Small-24B-2507", "ytb_asr_batch2": 0.0922, "average": 0.3753, "ytb_asr_batch1": 0.089, "mediacorp_short_test": 0.1151, "cna_test": 0.1612, "mediacorp_test": 0.4665, "idpc_short_test": 0.1779, "parliament_short_test": 0.0497, "ukusnews_short_test": 0.0602, "ukusnews_test": 0.9537, "idpc_test": 0.9685, "parliament_test": 0.9948, "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507" }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 0.4287, "cna_test": 0.273, "idpc_short_test": 0.719, "idpc_test": 1.165, "mediacorp_short_test": 0.188, "mediacorp_test": 0.4, "parliament_test": 0.194, "ukusnews_test": 0.176, "ytb_asr_batch1": 0.45, "ytb_asr_batch2": 0.904, "ukusnews_short_test": 0.127, "parliament_short_test": 0.1198 }, { "model": "gemma-3n-e4b-it", "idpc_short_test": 1.0054, "average": 0.4762, "ukusnews_short_test": 0.1048, "parliament_short_test": 0.0881, "mediacorp_test": 0.2299, "mediacorp_short_test": 0.3597, "ytb_asr_batch1": 0.3471, "ytb_asr_batch2": 0.1654, "cna_test": 2.5104, "parliament_test": 0.0895, "ukusnews_test": 0.0991, "idpc_test": 0.2389, "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it" }, { "model": "gemma-3n-e2b-it", "cna_test": 3.0407, "average": 0.4917, "parliament_short_test": 0.1222, "ukusnews_short_test": 0.1573, "idpc_test": 0.3001, "ukusnews_test": 0.1065, "parliament_test": 0.117, "idpc_short_test": 0.4866, "mediacorp_test": 0.2489, "mediacorp_short_test": 0.418, "ytb_asr_batch1": 0.2053, "ytb_asr_batch2": 0.2061, "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it" }, { "model": "Voxtral-Mini-3B-2507", "ytb_asr_batch2": 0.1491, "ytb_asr_batch1": 0.1143, "average": 0.5688, "mediacorp_short_test": 0.1625, "cna_test": 1.9728, "parliament_short_test": 0.0856, "ukusnews_short_test": 0.145, "mediacorp_test": 0.4834, "idpc_short_test": 0.2252, "ukusnews_test": 0.9523, "idpc_test": 0.9711, "parliament_test": 0.9953, "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507" }, { "model": "Omnilingual-LLM-ASR-7B", "modelLink": "https://arxiv.org/abs/2511.09690", "average": 0.9325, "cna_test": 1.048, "idpc_short_test": 1.007, "idpc_test": 0.916, "mediacorp_short_test": 1.033, "mediacorp_test": 0.944, "parliament_test": 0.853, "ukusnews_test": 0.934, "ytb_asr_batch1": 0.84, "ytb_asr_batch2": 0.784, "ukusnews_short_test": 0.9659 } ], "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "Speech Translation", "title": "Task: Speech Translation", "taskName": "st", "metric": "bleu", "metricInfo": "BLEU Score. The higher, the better. AudioBench columns are corpus BLEU; AudioBench-SEA columns are mean per-sample sentence-BLEU rescaled to 0-100. SeaLLMs-Audio-7B's Tamil translation columns reflect the same output-script limitation as its ASR-Tamil row: asked to translate into Tamil it answers in Thai, Latin or Chinese script, producing Tamil script on about 2% of predictions, so BLEU against Tamil references is near zero regardless of whether the content is right. Its non-Tamil targets in the same runs score normally. WavLLM's Tamil translation column reflects unstable generation rather than translation quality: it does produce Tamil script, but two thirds of those outputs collapse into a single repeated phrase (114 of 385 predictions on ytb_batch1_st_en_ta, against 3% on the Malay target in the same run). No repetition controls are configured for this model. Its remaining answers on that set are often in Chinese, the same output-language drift noted in its ASR-English row. MERaLiON-3-10B returns no output on 15 of 385 clips in ytb_batch1_st_en_ta (3.9%), and the score is averaged over all 385, so it is slightly understated against models that answer every clip. The blanks are specific audio clips, reproduced identically across two independent runs, not a transient failure. On the clips it does answer it produces Tamil script on 370 of 385 with no degenerate output, which is unusual in this column.", "ascending": false, "datasets": [ { "display": "CoVoST2-EN-ID", "internal": "covost2_en_id_test", "description": "CoVoST 2 dataset for speech translation from English to Indonesian.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_en_id_test_v1", "stats": { "num_rows": 15531, "audio_length": { "min": 1.1, "max": 142.54, "mean": 5.71, "median": 5.42, "std": 2.72, "total_hours": 24.65, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 15524, 6, 1, 0, 0 ] } } } }, { "display": "CoVoST2-EN-ZH", "internal": "covost2_en_zh_test", "description": "CoVoST 2 dataset for speech translation from English to Chinese.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_en_zh_test_v1", "stats": { "num_rows": 15531, "audio_length": { "min": 1.1, "max": 142.54, "mean": 5.71, "median": 5.42, "std": 2.72, "total_hours": 24.65, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 15524, 6, 1, 0, 0 ] } } } }, { "display": "CoVoST2-EN-TA", "internal": "covost2_en_ta_test", "description": "CoVoST 2 dataset for speech translation from English to Tamil.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_en_ta_test_v1", "stats": { "num_rows": 15531, "audio_length": { "min": 1.1, "max": 142.54, "mean": 5.71, "median": 5.42, "std": 2.72, "total_hours": 24.65, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 15524, 6, 1, 0, 0 ] } } } }, { "display": "CoVoST2-ID-EN", "internal": "covost2_id_en_test", "description": "CoVoST 2 dataset for speech translation from Indonesian to English.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_id_en_test_v1", "stats": { "num_rows": 844, "audio_length": { "min": 1.68, "max": 8.02, "mean": 3.89, "median": 3.7, "std": 1.06, "total_hours": 0.91, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 844, 0, 0, 0, 0 ] } } } }, { "display": "CoVoST2-ZH-EN", "internal": "covost2_zh_en_test", "description": "CoVoST 2 dataset for speech translation from Chinese to English.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_zh_en_test_v1", "stats": { "num_rows": 4898, "audio_length": { "min": 1.75, "max": 10.58, "mean": 6.06, "median": 5.9, "std": 2.03, "total_hours": 8.24, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 4898, 0, 0, 0, 0 ] } } } }, { "display": "CoVoST2-TA-EN", "internal": "covost2_ta_en_test", "description": "CoVoST 2 dataset for speech translation from Tamil to English.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_ta_en_test_v2", "stats": { "num_rows": 754, "audio_length": { "min": 1.94, "max": 9.86, "mean": 4.71, "median": 4.58, "std": 1.18, "total_hours": 0.99, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 754, 0, 0, 0, 0 ] } } } }, { "display": "Fleurs-ID-30 [SEA]", "internal": "translation_fleurs_id_30", "description": "FLEURS Indonesian → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "stats": {} }, { "display": "Fleurs-KM-30 [SEA]", "internal": "translation_fleurs_km_30", "description": "FLEURS Khmer → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "stats": {} }, { "display": "Fleurs-LO-30 [SEA]", "internal": "translation_fleurs_lo_30", "description": "FLEURS Lao → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "stats": {} }, { "display": "Fleurs-MS-30 [SEA]", "internal": "translation_fleurs_ms_30", "description": "FLEURS Malay → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "stats": {} }, { "display": "Fleurs-MY-30 [SEA]", "internal": "translation_fleurs_my_30", "description": "FLEURS Burmese → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "stats": {} }, { "display": "Fleurs-TH-30 [SEA]", "internal": "translation_fleurs_th_30", "description": "FLEURS Thai → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "stats": {} }, { "display": "Fleurs-TL-30 [SEA]", "internal": "translation_fleurs_tl_30", "description": "FLEURS Tagalog → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "stats": {} }, { "display": "Fleurs-VI-30 [SEA]", "internal": "translation_fleurs_vi_30", "description": "FLEURS Vietnamese → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "stats": {} }, { "display": "Fleurs-ZH-30 [SEA]", "internal": "translation_fleurs_zh_30", "description": "FLEURS Chinese → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "stats": {} }, { "display": "YTB-Batch1-ST-EN-MS", "internal": "ytb_batch1_st_en_ms", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "stats": {} }, { "display": "YTB-Batch1-ST-EN-TA", "internal": "ytb_batch1_st_en_ta", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "stats": {} }, { "display": "YTB-Batch1-ST-EN-ZH", "internal": "ytb_batch1_st_en_zh", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "stats": {} } ], "data": { "columns": [ { "key": "covost2_en_id_test", "display": "CoVoST2-EN-ID", "description": "CoVoST 2 dataset for speech translation from English to Indonesian.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_en_id_test_v1", "isWer": false }, { "key": "covost2_en_zh_test", "display": "CoVoST2-EN-ZH", "description": "CoVoST 2 dataset for speech translation from English to Chinese.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_en_zh_test_v1", "isWer": false }, { "key": "covost2_en_ta_test", "display": "CoVoST2-EN-TA", "description": "CoVoST 2 dataset for speech translation from English to Tamil.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_en_ta_test_v1", "isWer": false }, { "key": "covost2_id_en_test", "display": "CoVoST2-ID-EN", "description": "CoVoST 2 dataset for speech translation from Indonesian to English.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_id_en_test_v1", "isWer": false }, { "key": "covost2_zh_en_test", "display": "CoVoST2-ZH-EN", "description": "CoVoST 2 dataset for speech translation from Chinese to English.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_zh_en_test_v1", "isWer": false }, { "key": "covost2_ta_en_test", "display": "CoVoST2-TA-EN", "description": "CoVoST 2 dataset for speech translation from Tamil to English.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/covost2_ta_en_test_v2", "isWer": false }, { "key": "translation_fleurs_id_30", "display": "Fleurs-ID-30 [SEA]", "description": "FLEURS Indonesian → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "isWer": false }, { "key": "translation_fleurs_km_30", "display": "Fleurs-KM-30 [SEA]", "description": "FLEURS Khmer → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "isWer": false }, { "key": "translation_fleurs_lo_30", "display": "Fleurs-LO-30 [SEA]", "description": "FLEURS Lao → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "isWer": false }, { "key": "translation_fleurs_ms_30", "display": "Fleurs-MS-30 [SEA]", "description": "FLEURS Malay → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "isWer": false }, { "key": "translation_fleurs_my_30", "display": "Fleurs-MY-30 [SEA]", "description": "FLEURS Burmese → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "isWer": false }, { "key": "translation_fleurs_th_30", "display": "Fleurs-TH-30 [SEA]", "description": "FLEURS Thai → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "isWer": false }, { "key": "translation_fleurs_tl_30", "display": "Fleurs-TL-30 [SEA]", "description": "FLEURS Tagalog → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "isWer": false }, { "key": "translation_fleurs_vi_30", "display": "Fleurs-VI-30 [SEA]", "description": "FLEURS Vietnamese → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "isWer": false }, { "key": "translation_fleurs_zh_30", "display": "Fleurs-ZH-30 [SEA]", "description": "FLEURS Chinese → English ST (30s) (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ST", "isWer": false }, { "key": "ytb_batch1_st_en_ms", "display": "YTB-Batch1-ST-EN-MS", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "isWer": false }, { "key": "ytb_batch1_st_en_ta", "display": "YTB-Batch1-ST-EN-TA", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "isWer": false }, { "key": "ytb_batch1_st_en_zh", "display": "YTB-Batch1-ST-EN-ZH", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "translation_fleurs_id_30": 34.3788, "translation_fleurs_km_30": 1.4662, "translation_fleurs_lo_30": 12.5441, "translation_fleurs_ms_30": 27.7484, "translation_fleurs_my_30": 1.1092, "translation_fleurs_th_30": 20.3346, "translation_fleurs_tl_30": 21.624, "translation_fleurs_vi_30": 24.053, "translation_fleurs_zh_30": 23.0233, "average": 20.4923, "covost2_en_id_test": 30.546, "covost2_en_ta_test": 9.0546, "covost2_en_zh_test": 43.0004, "covost2_id_en_test": 44.3384, "ytb_batch1_st_en_ms": 21.1504, "ytb_batch1_st_en_ta": 1.8375, "covost2_zh_en_test": 20.5375, "covost2_ta_en_test": 1.7106, "ytb_batch1_st_en_zh": 30.4048 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "translation_fleurs_zh_30": 0.7203, "average": 19.6952, "translation_fleurs_id_30": 33.8634, "translation_fleurs_km_30": 5.903, "covost2_en_id_test": 25.7491, "covost2_zh_en_test": 15.7429, "translation_fleurs_my_30": 1.3216, "covost2_en_zh_test": 41.4988, "covost2_id_en_test": 43.9637, "ytb_batch1_st_en_zh": 32.8371, "translation_fleurs_th_30": 17.0318, "translation_fleurs_vi_30": 22.6495, "covost2_ta_en_test": 3.7168, "ytb_batch1_st_en_ta": 5.5278, "translation_fleurs_lo_30": 9.0733, "translation_fleurs_ms_30": 32.6717, "translation_fleurs_tl_30": 29.5587, "covost2_en_ta_test": 9.2197, "ytb_batch1_st_en_ms": 23.4638 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 18.2969, "covost2_en_id_test": 29.0497, "covost2_en_zh_test": 39.3309, "covost2_en_ta_test": 8.8227, "covost2_id_en_test": 40.1228, "covost2_zh_en_test": 15.8062, "covost2_ta_en_test": 4.7455, "translation_fleurs_id_30": 25.3551, "translation_fleurs_km_30": 1.7857, "translation_fleurs_lo_30": 8.5873, "translation_fleurs_ms_30": 25.5009, "translation_fleurs_my_30": 0.6243, "translation_fleurs_th_30": 12.2625, "translation_fleurs_tl_30": 21.0086, "translation_fleurs_vi_30": 14.7512, "translation_fleurs_zh_30": 14.883, "ytb_batch1_st_en_ms": 27.9315, "ytb_batch1_st_en_ta": 5.385, "ytb_batch1_st_en_zh": 33.3905 }, { "model": "MERaLiON-3-10B", "translation_fleurs_id_30": 26.7715, "translation_fleurs_km_30": 2.0443, "translation_fleurs_lo_30": 1.9855, "translation_fleurs_ms_30": 23.2454, "translation_fleurs_my_30": 0.4505, "translation_fleurs_th_30": 5.3753, "translation_fleurs_tl_30": 19.0937, "translation_fleurs_vi_30": 17.132, "translation_fleurs_zh_30": 15.3004, "covost2_en_id_test": 29.3583, "covost2_en_ta_test": 10.2384, "covost2_en_zh_test": 39.1112, "covost2_id_en_test": 39.8582, "covost2_ta_en_test": 4.6557, "covost2_zh_en_test": 15.1439, "ytb_batch1_st_en_ms": 27.5184, "ytb_batch1_st_en_zh": 32.8527, "average": 17.5203, "ytb_batch1_st_en_ta": 5.2296, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 16.6073, "covost2_en_id_test": 29.1611, "covost2_en_zh_test": 39.4222, "covost2_en_ta_test": 10.9478, "covost2_id_en_test": 37.5984, "covost2_zh_en_test": 15.038, "covost2_ta_en_test": 5.8768, "translation_fleurs_id_30": 21.4911, "translation_fleurs_km_30": 2.076, "translation_fleurs_lo_30": 5.4825, "translation_fleurs_ms_30": 16.6481, "translation_fleurs_my_30": 1.1082, "translation_fleurs_th_30": 9.5229, "translation_fleurs_tl_30": 16.9998, "translation_fleurs_vi_30": 12.544, "translation_fleurs_zh_30": 13.5234, "ytb_batch1_st_en_ms": 24.3756, "ytb_batch1_st_en_ta": 6.0614, "ytb_batch1_st_en_zh": 31.0538 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 16.4481, "translation_fleurs_my_30": 1.1254, "covost2_en_id_test": 17.7487, "covost2_id_en_test": 40.6573, "covost2_zh_en_test": 12.389, "ytb_batch1_st_en_zh": 28.0541, "translation_fleurs_km_30": 4.8643, "translation_fleurs_vi_30": 19.2295, "covost2_en_zh_test": 32.5782, "ytb_batch1_st_en_ta": 2.325, "translation_fleurs_ms_30": 24.6443, "translation_fleurs_th_30": 13.9298, "covost2_ta_en_test": 4.0243, "translation_fleurs_tl_30": 23.4816, "covost2_en_ta_test": 4.7543, "ytb_batch1_st_en_ms": 14.7795, "translation_fleurs_id_30": 27.7948, "translation_fleurs_lo_30": 8.1434, "translation_fleurs_zh_30": 15.5426 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 16.1318, "covost2_en_id_test": 29.5563, "covost2_en_zh_test": 40.95, "covost2_en_ta_test": 9.7775, "covost2_id_en_test": 40.5774, "covost2_zh_en_test": 15.6474, "covost2_ta_en_test": 6.9411, "translation_fleurs_id_30": 17.9774, "translation_fleurs_km_30": 1.2694, "translation_fleurs_lo_30": 4.1934, "translation_fleurs_ms_30": 18.092, "translation_fleurs_my_30": 0.3974, "translation_fleurs_th_30": 3.8808, "translation_fleurs_tl_30": 15.3434, "translation_fleurs_vi_30": 9.9752, "translation_fleurs_zh_30": 13.2029, "ytb_batch1_st_en_ms": 25.1675, "ytb_batch1_st_en_ta": 5.5004, "ytb_batch1_st_en_zh": 31.9221 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 14.7437, "covost2_en_id_test": 19.8306, "covost2_en_zh_test": 37.4877, "covost2_en_ta_test": 1.9266, "covost2_id_en_test": 36.2327, "covost2_zh_en_test": 21.2505, "covost2_ta_en_test": 0.734, "translation_fleurs_id_30": 28.7464, "translation_fleurs_km_30": 1.2462, "translation_fleurs_lo_30": 6.0249, "translation_fleurs_ms_30": 20.1824, "translation_fleurs_my_30": 1.2601, "translation_fleurs_th_30": 14.1888, "translation_fleurs_tl_30": 4.8758, "translation_fleurs_vi_30": 18.3552, "translation_fleurs_zh_30": 20.4276, "ytb_batch1_st_en_ms": 7.2546, "ytb_batch1_st_en_ta": 0.1292, "ytb_batch1_st_en_zh": 25.234 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 13.3731, "covost2_en_id_test": 17.5695, "covost2_en_zh_test": 36.8719, "covost2_en_ta_test": 0.8422, "covost2_id_en_test": 35.7828, "covost2_zh_en_test": 20.1909, "covost2_ta_en_test": 0.9786, "translation_fleurs_id_30": 25.5883, "translation_fleurs_km_30": 1.2889, "translation_fleurs_lo_30": 5.1905, "translation_fleurs_ms_30": 18.649, "translation_fleurs_my_30": 1.2021, "translation_fleurs_th_30": 12.0063, "translation_fleurs_tl_30": 5.2245, "translation_fleurs_vi_30": 16.2761, "translation_fleurs_zh_30": 19.3503, "ytb_batch1_st_en_ms": 0.9897, "ytb_batch1_st_en_ta": 0.0906, "ytb_batch1_st_en_zh": 22.6244 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 11.9924, "covost2_en_id_test": 24.1158, "covost2_en_zh_test": 35.5346, "covost2_en_ta_test": 5.399, "covost2_id_en_test": 31.4328, "covost2_zh_en_test": 11.4779, "covost2_ta_en_test": 4.3342, "translation_fleurs_id_30": 15.7013, "translation_fleurs_km_30": 0.7867, "translation_fleurs_lo_30": 1.4625, "translation_fleurs_ms_30": 12.3013, "translation_fleurs_my_30": 0.5686, "translation_fleurs_th_30": 2.3799, "translation_fleurs_tl_30": 6.3511, "translation_fleurs_vi_30": 5.0384, "translation_fleurs_zh_30": 8.1427, "ytb_batch1_st_en_ms": 22.1597, "ytb_batch1_st_en_ta": 1.2701, "ytb_batch1_st_en_zh": 27.4068 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 11.8488, "covost2_en_id_test": 22.0603, "covost2_en_zh_test": 31.3587, "covost2_en_ta_test": 0.007, "covost2_id_en_test": 38.5943, "covost2_zh_en_test": 14.4936, "covost2_ta_en_test": 0.7611, "translation_fleurs_id_30": 20.7533, "translation_fleurs_km_30": 1.1597, "translation_fleurs_lo_30": 4.5903, "translation_fleurs_ms_30": 15.6705, "translation_fleurs_my_30": 1.0438, "translation_fleurs_th_30": 8.6845, "translation_fleurs_tl_30": 1.8053, "translation_fleurs_vi_30": 10.1677, "translation_fleurs_zh_30": 12.76, "ytb_batch1_st_en_ms": 8.9111, "ytb_batch1_st_en_zh": 20.4528, "ytb_batch1_st_en_ta": 0.0051 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "translation_fleurs_id_30": 16.3355, "translation_fleurs_km_30": 4.0165, "translation_fleurs_lo_30": 7.3028, "translation_fleurs_ms_30": 14.872, "translation_fleurs_my_30": 1.052, "translation_fleurs_th_30": 11.5305, "translation_fleurs_tl_30": 15.3572, "translation_fleurs_vi_30": 6.1944, "translation_fleurs_zh_30": 6.1974, "average": 11.6336, "covost2_en_id_test": 16.7696, "covost2_en_ta_test": 6.1468, "covost2_en_zh_test": 23.9755, "covost2_id_en_test": 22.1785, "covost2_ta_en_test": 1.9099, "ytb_batch1_st_en_ms": 16.8042, "ytb_batch1_st_en_ta": 5.9615, "ytb_batch1_st_en_zh": 27.2599, "covost2_zh_en_test": 5.5414 }, { "model": "MERaLiON-3-3B-ASR", "covost2_en_id_test": 25.3938, "average": 10.8311, "covost2_en_ta_test": 7.9855, "covost2_en_zh_test": 36.5046, "translation_fleurs_id_30": 8.856, "translation_fleurs_km_30": 0.7509, "translation_fleurs_lo_30": 0.2115, "translation_fleurs_ms_30": 3.2566, "translation_fleurs_my_30": 0.9623, "covost2_id_en_test": 28.9586, "covost2_zh_en_test": 10.8432, "covost2_ta_en_test": 4.7563, "translation_fleurs_th_30": 0.0967, "translation_fleurs_tl_30": 3.2398, "translation_fleurs_vi_30": 0.293, "translation_fleurs_zh_30": 7.247, "ytb_batch1_st_en_ms": 23.2873, "ytb_batch1_st_en_ta": 2.6275, "ytb_batch1_st_en_zh": 29.6889, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-3-3B-ASR-CTM", "covost2_en_id_test": 25.4515, "average": 10.7081, "covost2_en_ta_test": 7.9842, "covost2_en_zh_test": 35.9625, "translation_fleurs_id_30": 5.499, "translation_fleurs_km_30": 1.0119, "translation_fleurs_lo_30": 1.0884, "translation_fleurs_ms_30": 6.5317, "translation_fleurs_my_30": 1.0359, "covost2_id_en_test": 22.5432, "covost2_zh_en_test": 7.7977, "covost2_ta_en_test": 4.4439, "translation_fleurs_th_30": 2.8596, "translation_fleurs_tl_30": 1.9308, "translation_fleurs_vi_30": 2.2751, "translation_fleurs_zh_30": 9.6678, "ytb_batch1_st_en_ms": 23.999, "ytb_batch1_st_en_ta": 2.6524, "ytb_batch1_st_en_zh": 30.0113 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "translation_fleurs_id_30": 22.8477, "translation_fleurs_km_30": 1.6274, "translation_fleurs_lo_30": 7.1227, "translation_fleurs_ms_30": 19.8869, "translation_fleurs_my_30": 0.7067, "translation_fleurs_th_30": 11.5496, "translation_fleurs_tl_30": 11.5169, "translation_fleurs_vi_30": 9.4004, "translation_fleurs_zh_30": 7.5962, "average": 10.2558, "covost2_en_id_test": 14.4506, "covost2_en_ta_test": 4.2931, "covost2_en_zh_test": 16.6231, "covost2_id_en_test": 27.1848, "ytb_batch1_st_en_ms": 7.6032, "ytb_batch1_st_en_ta": 2.4601, "covost2_zh_en_test": 6.4272, "covost2_ta_en_test": 1.4247, "ytb_batch1_st_en_zh": 11.8832 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "average": 9.3736, "covost2_en_id_test": 0.8409, "covost2_en_zh_test": 2.8035, "covost2_en_ta_test": 0.0029, "covost2_id_en_test": 38.7402, "covost2_zh_en_test": 9.0925, "covost2_ta_en_test": 2.4624, "translation_fleurs_id_30": 22.8748, "translation_fleurs_km_30": 4.2808, "translation_fleurs_lo_30": 6.8074, "translation_fleurs_ms_30": 21.2934, "translation_fleurs_my_30": 0.7647, "translation_fleurs_th_30": 11.2352, "translation_fleurs_tl_30": 17.9194, "translation_fleurs_vi_30": 15.2974, "translation_fleurs_zh_30": 12.5644, "ytb_batch1_st_en_ms": 0.3023, "ytb_batch1_st_en_ta": 0.0344, "ytb_batch1_st_en_zh": 1.4088 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "translation_fleurs_id_30": 16.5178, "translation_fleurs_km_30": 2.5581, "translation_fleurs_lo_30": 5.4985, "translation_fleurs_ms_30": 13.4793, "translation_fleurs_my_30": 0.8461, "translation_fleurs_th_30": 8.086, "translation_fleurs_tl_30": 10.5167, "translation_fleurs_vi_30": 4.2938, "translation_fleurs_zh_30": 4.2524, "average": 8.8443, "covost2_en_id_test": 12.4748, "covost2_en_ta_test": 4.0511, "covost2_en_zh_test": 18.3203, "covost2_id_en_test": 10.663, "covost2_ta_en_test": 1.2952, "ytb_batch1_st_en_ms": 13.9395, "ytb_batch1_st_en_ta": 4.4926, "ytb_batch1_st_en_zh": 24.5126, "covost2_zh_en_test": 3.4001 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "translation_fleurs_id_30": 19.9757, "translation_fleurs_km_30": 1.6657, "translation_fleurs_lo_30": 4.6382, "translation_fleurs_ms_30": 17.3314, "translation_fleurs_my_30": 1.029, "translation_fleurs_th_30": 7.6459, "translation_fleurs_tl_30": 7.8212, "translation_fleurs_vi_30": 6.6415, "translation_fleurs_zh_30": 4.4194, "covost2_en_id_test": 10.587, "covost2_en_ta_test": 2.6483, "covost2_en_zh_test": 9.8501, "covost2_id_en_test": 25.8356, "covost2_ta_en_test": 0.9411, "covost2_zh_en_test": 3.8971, "ytb_batch1_st_en_ms": 14.0144, "ytb_batch1_st_en_ta": 2.8829, "ytb_batch1_st_en_zh": 12.7623, "average": 8.5882, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 7.5867, "covost2_en_id_test": 7.8437, "covost2_en_zh_test": 28.6175, "covost2_en_ta_test": 0.0018, "covost2_id_en_test": 22.5869, "covost2_zh_en_test": 4.7267, "covost2_ta_en_test": 2.2809, "translation_fleurs_id_30": 13.1945, "translation_fleurs_km_30": 1.925, "translation_fleurs_lo_30": 2.8669, "translation_fleurs_ms_30": 11.8204, "translation_fleurs_my_30": 0.6516, "translation_fleurs_th_30": 5.6417, "translation_fleurs_tl_30": 10.1305, "translation_fleurs_vi_30": 8.4117, "translation_fleurs_zh_30": 6.1453, "ytb_batch1_st_en_ms": 1.23, "ytb_batch1_st_en_ta": 0.0498, "ytb_batch1_st_en_zh": 8.4352 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 6.793, "covost2_en_id_test": 13.1716, "covost2_en_zh_test": 40.9786, "covost2_en_ta_test": 0.401, "covost2_id_en_test": 1.1655, "covost2_zh_en_test": 13.8336, "covost2_ta_en_test": 0.4891, "ytb_batch1_st_en_ms": 8.5777, "ytb_batch1_st_en_ta": 0.3386, "ytb_batch1_st_en_zh": 23.472, "translation_fleurs_id_30": 1.1769, "translation_fleurs_km_30": 0.8259, "translation_fleurs_lo_30": 0.7024, "translation_fleurs_ms_30": 1.2591, "translation_fleurs_my_30": 0.7502, "translation_fleurs_th_30": 0.7488, "translation_fleurs_tl_30": 1.7507, "translation_fleurs_vi_30": 0.7799, "translation_fleurs_zh_30": 11.8532 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 6.3145, "covost2_en_id_test": 14.2293, "covost2_en_zh_test": 20.9163, "covost2_en_ta_test": 0.3962, "covost2_id_en_test": 5.3385, "covost2_zh_en_test": 11.7952, "covost2_ta_en_test": 1.1219, "translation_fleurs_id_30": 5.2488, "translation_fleurs_km_30": 0.876, "translation_fleurs_lo_30": 0.7678, "translation_fleurs_ms_30": 3.2307, "translation_fleurs_my_30": 0.89, "translation_fleurs_th_30": 0.808, "translation_fleurs_tl_30": 1.2529, "translation_fleurs_vi_30": 1.8703, "translation_fleurs_zh_30": 13.1063, "ytb_batch1_st_en_ms": 8.3169, "ytb_batch1_st_en_ta": 0.1465, "ytb_batch1_st_en_zh": 23.3497 }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 6.1711, "covost2_en_id_test": 11.7038, "covost2_en_zh_test": 30.3291, "covost2_en_ta_test": 0.0107, "covost2_id_en_test": 6.0737, "covost2_zh_en_test": 1.9696, "covost2_ta_en_test": 0.5429, "translation_fleurs_id_30": 7.1279, "translation_fleurs_km_30": 1.2708, "translation_fleurs_lo_30": 2.0565, "translation_fleurs_ms_30": 6.1429, "translation_fleurs_my_30": 0.235, "translation_fleurs_tl_30": 5.8387, "translation_fleurs_vi_30": 5.1463, "translation_fleurs_th_30": 3.477, "translation_fleurs_zh_30": 3.6507, "ytb_batch1_st_en_ms": 0.4947, "ytb_batch1_st_en_zh": 18.8388 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 2.8186, "covost2_en_id_test": 8.5438, "covost2_en_zh_test": 2.5162, "covost2_en_ta_test": 0.3184, "covost2_id_en_test": 0.9738, "covost2_zh_en_test": 14.6972, "covost2_ta_en_test": 0.6041, "translation_fleurs_lo_30": 0.5306, "translation_fleurs_id_30": 0.6527, "translation_fleurs_km_30": 0.5599, "translation_fleurs_ms_30": 0.8233, "translation_fleurs_my_30": 0.6055, "translation_fleurs_th_30": 0.608, "translation_fleurs_tl_30": 0.9217, "translation_fleurs_vi_30": 0.6241, "ytb_batch1_st_en_ta": 0.0416, "ytb_batch1_st_en_zh": 1.659, "translation_fleurs_zh_30": 10.5681, "ytb_batch1_st_en_ms": 5.4872 } ], "metric": "bleu", "metricInfo": "BLEU Score. The higher, the better. AudioBench columns are corpus BLEU; AudioBench-SEA columns are mean per-sample sentence-BLEU rescaled to 0-100. SeaLLMs-Audio-7B's Tamil translation columns reflect the same output-script limitation as its ASR-Tamil row: asked to translate into Tamil it answers in Thai, Latin or Chinese script, producing Tamil script on about 2% of predictions, so BLEU against Tamil references is near zero regardless of whether the content is right. Its non-Tamil targets in the same runs score normally. WavLLM's Tamil translation column reflects unstable generation rather than translation quality: it does produce Tamil script, but two thirds of those outputs collapse into a single repeated phrase (114 of 385 predictions on ytb_batch1_st_en_ta, against 3% on the Malay target in the same run). No repetition controls are configured for this model. Its remaining answers on that set are often in Chinese, the same output-language drift noted in its ASR-English row. MERaLiON-3-10B returns no output on 15 of 385 clips in ytb_batch1_st_en_ta (3.9%), and the score is averaged over all 385, so it is slightly understated against models that answer every clip. The blanks are specific audio clips, reproduced identically across two independent runs, not a transient failure. On the clips it does answer it produces Tamil script on 370 of 385 with no degenerate output, which is unusual in this column.", "ascending": false } }, { "key": "SQA-English", "title": "Task: Spoken Question Answering - English", "taskName": "sqa_english", "metric": "judge", "metricInfo": "Model-as-a-Judge performance. Scale from 0-100. The higher, the better. AudioBench columns are judged by LLaMA-3-70B; AudioBench-SEA columns use the AudioBench-SEA judge.", "ascending": false, "datasets": [ { "display": "CN-College-Listen-MCQ", "internal": "cn_college_listen_mcq_test", "description": "Chinese College English Listening Test, with multiple-choice questions.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/cn_college_listen_mcq_test", "stats": { "num_rows": 2271, "audio_length": { "min": 5.76, "max": 137.82, "mean": 21.09, "median": 17.65, "std": 15.17, "total_hours": 13.3, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 2082, 181, 8, 0, 0 ] } } } }, { "display": "DREAM-TTS-MCQ", "internal": "dream_tts_mcq_test", "description": "DREAM dataset for spoken question-answering, derived from textual data and synthesized speech.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/dream_tts_mcq_test", "stats": { "num_rows": 1913, "audio_length": { "min": 3.15, "max": 261.86, "mean": 34.14, "median": 24.83, "std": 32.38, "total_hours": 18.14, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 1021, 867, 25, 0, 0 ] } } } }, { "display": "SLUE-P2-SQA5", "internal": "slue_p2_sqa5_test", "description": "Spoken Language Understanding Evaluation (SLUE) dataset, part 2, focused on QA tasks.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/slue_p2_sqa5_test", "stats": { "num_rows": 408, "audio_length": { "min": 13.01, "max": 40.0, "mean": 39.86, "median": 40.0, "std": 1.66, "total_hours": 4.52, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 2, 406, 0, 0, 0 ] } } } }, { "display": "Public-SG-Speech-QA", "internal": "public_sg_speech_qa_test", "description": "Public dataset for speech-based question answering, gathered from Singapore.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/public_sg_speech_qa_test", "stats": { "num_rows": 688, "audio_length": { "min": 15.79, "max": 95.71, "mean": 39.86, "median": 39.38, "std": 10.55, "total_hours": 7.62, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 104, 584, 0, 0, 0 ] } } } }, { "display": "Spoken-SQuAD", "internal": "spoken_squad_test", "description": "Spoken SQuAD dataset, based on the textual SQuAD dataset, converted into audio.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/spoken_squad_test_v1", "stats": { "num_rows": 5351, "audio_length": { "min": 11.57, "max": 305.3, "mean": 61.05, "median": 55.39, "std": 28.31, "total_hours": 90.74, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 381, 4795, 172, 3, 0 ] } } } }, { "display": "MMAU-mini", "internal": "mmau_mini", "description": "A compact audio-language benchmark subset used in AudioBench for speech understanding evaluation, providing lightweight but diverse test samples.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/MMAU-mini", "stats": null }, { "display": "Yodas2-EN-30 [SEA]", "internal": "sqa_yodas2_en_30", "description": "YODAS2 English SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "stats": {} }, { "display": "YTB-SQA-Batch1", "internal": "ytb_sqa_batch1", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "stats": {} } ], "data": { "columns": [ { "key": "slue_p2_sqa5_test", "display": "SLUE-P2-SQA5", "description": "Spoken Language Understanding Evaluation (SLUE) dataset, part 2, focused on QA tasks.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/slue_p2_sqa5_test", "isWer": false }, { "key": "public_sg_speech_qa_test", "display": "Public-SG-Speech-QA", "description": "Public dataset for speech-based question answering, gathered from Singapore.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/public_sg_speech_qa_test", "isWer": false }, { "key": "spoken_squad_test", "display": "Spoken-SQuAD", "description": "Spoken SQuAD dataset, based on the textual SQuAD dataset, converted into audio.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/spoken_squad_test_v1", "isWer": false }, { "key": "cn_college_listen_mcq_test", "display": "CN-College-Listen-MCQ", "description": "Chinese College English Listening Test, with multiple-choice questions.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/cn_college_listen_mcq_test", "isWer": false }, { "key": "dream_tts_mcq_test", "display": "DREAM-TTS-MCQ", "description": "DREAM dataset for spoken question-answering, derived from textual data and synthesized speech.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/dream_tts_mcq_test", "isWer": false }, { "key": "mmau_mini", "display": "MMAU-mini", "description": "A compact audio-language benchmark subset used in AudioBench for speech understanding evaluation, providing lightweight but diverse test samples.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/MMAU-mini", "isWer": false }, { "key": "sqa_yodas2_en_30", "display": "Yodas2-EN-30 [SEA]", "description": "YODAS2 English SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "isWer": false }, { "key": "ytb_sqa_batch1", "display": "YTB-SQA-Batch1", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/audiobench_datasets", "isWer": false } ], "rows": [ { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 73.9758, "slue_p2_sqa5_test": 84.853, "public_sg_speech_qa_test": 71.744, "spoken_squad_test": 84.717, "cn_college_listen_mcq_test": 72.655, "dream_tts_mcq_test": 74.386, "mmau_mini": 55.5 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "sqa_yodas2_en_30": 86.20967742, "average": 72.8008, "cn_college_listen_mcq_test": 87.0101277, "dream_tts_mcq_test": 65.62467329, "mmau_mini": 64.48, "public_sg_speech_qa_test": 62.96511628, "slue_p2_sqa5_test": 66.8627451, "ytb_sqa_batch1": 76.45320197 }, { "model": "MERaLiON-3-10B", "cn_college_listen_mcq_test": 82.21928666, "public_sg_speech_qa_test": 62.52906977, "slue_p2_sqa5_test": 65.68627451, "spoken_squad_test": 57.24911232, "ytb_sqa_batch1": 75.68472906, "average": 69.302, "audiocaps_qa_test": 52.07667732, "imda_part3_30s_sqa_human_test": 66.2, "imda_part4_30s_sqa_human_test": 65.4, "imda_part5_30s_sqa_human_test": 75.4, "imda_part6_30s_sqa_human_test": 72.6, "sqa_yodas2_en_30": 86.12903226, "dream_tts_mcq_test": 65.25875588, "mmau_mini": 59.66, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 67.293, "slue_p2_sqa5_test": 68.52941176, "public_sg_speech_qa_test": 64.59302326, "spoken_squad_test": 56.72958326, "cn_college_listen_mcq_test": 75.59665346, "dream_tts_mcq_test": 61.30684788, "mmau_mini": 53.28, "sqa_yodas2_en_30": 85.34274194, "ytb_sqa_batch1": 72.96551724 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 64.6868, "slue_p2_sqa5_test": 64.75490196, "public_sg_speech_qa_test": 64.30232558, "spoken_squad_test": 52.95832555, "cn_college_listen_mcq_test": 67.07177455, "dream_tts_mcq_test": 60.65865133, "mmau_mini": 57.88, "sqa_yodas2_en_30": 81.57258065, "ytb_sqa_batch1": 68.2955665 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 63.2918, "slue_p2_sqa5_test": 61.8627451, "public_sg_speech_qa_test": 60.58139535, "spoken_squad_test": 51.84825266, "cn_college_listen_mcq_test": 75.55261999, "dream_tts_mcq_test": 59.10088866, "mmau_mini": 53.68, "sqa_yodas2_en_30": 76.69354839, "ytb_sqa_batch1": 67.01477833 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 62.5927, "slue_p2_sqa5_test": 83.725, "public_sg_speech_qa_test": 74.186, "spoken_squad_test": 83.196, "cn_college_listen_mcq_test": 75.649, "dream_tts_mcq_test": 0.0, "mmau_mini": 58.8 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 59.7612, "slue_p2_sqa5_test": 64.50980392, "public_sg_speech_qa_test": 63.19767442, "spoken_squad_test": 51.94916838, "cn_college_listen_mcq_test": 59.04007045, "dream_tts_mcq_test": 48.09200209, "mmau_mini": 45.1, "sqa_yodas2_en_30": 77.11693548, "ytb_sqa_batch1": 69.08374384 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 58.894, "slue_p2_sqa5_test": 52.5, "public_sg_speech_qa_test": 56.25, "spoken_squad_test": 48.08073257, "cn_college_listen_mcq_test": 70.45354469, "dream_tts_mcq_test": 55.06534239, "mmau_mini": 52.04, "ytb_sqa_batch1": 63.13300493, "sqa_yodas2_en_30": 73.62903226 }, { "model": "MERaLiON-3-3B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR", "sqa_yodas2_en_30": 69.19354839, "audiocaps_qa_test": 43.64217252, "dream_tts_mcq_test": 45.01829587, "imda_part3_30s_sqa_human_test": 58.4, "imda_part5_30s_sqa_human_test": 65.4, "openhermes_audio_test": 20.6, "spoken_squad_test": 52.5004672, "ytb_sqa_batch1": 67.33004926, "average": 58.5106 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "sqa_yodas2_en_30": 58.36693548, "cn_college_listen_mcq_test": 54.47820343, "imda_part3_30s_sqa_human_test": 38.8, "imda_part5_30s_sqa_human_test": 45.6, "slue_p2_sqa5_test": 43.92156863, "ytb_sqa_batch1": 64.86699507, "average": 49.2543, "audiocaps_qa_test": 20.06389776, "dream_tts_mcq_test": 42.27914271, "imda_part4_30s_sqa_human_test": 44.8, "imda_part6_30s_sqa_human_test": 45.6, "mmau_mini": 23.92, "public_sg_speech_qa_test": 57.87790698, "spoken_squad_test": 48.32367782, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT", "alpaca_audio_test": 64.2, "openhermes_audio_test": 60.0 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 45.9956, "slue_p2_sqa5_test": 82.843, "public_sg_speech_qa_test": 0.0, "spoken_squad_test": 67.094, "cn_college_listen_mcq_test": 81.726, "dream_tts_mcq_test": 75.902, "mmau_mini": 60.4, "sqa_yodas2_en_30": 0.0, "ytb_sqa_batch1": 0.0 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 31.9461, "slue_p2_sqa5_test": 0.0, "public_sg_speech_qa_test": 0.0, "spoken_squad_test": 65.648, "cn_college_listen_mcq_test": 50.815, "dream_tts_mcq_test": 56.56, "mmau_mini": 50.6, "ytb_sqa_batch1": 0.0 } ], "metric": "judge", "metricInfo": "Model-as-a-Judge performance. Scale from 0-100. The higher, the better. AudioBench columns are judged by LLaMA-3-70B; AudioBench-SEA columns use the AudioBench-SEA judge.", "ascending": false } }, { "key": "SQA-Singlish", "title": "Task: Spoken Question Answering - Singlish", "taskName": "sqa_singlish", "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false, "datasets": [ { "display": "MNSC-PART3-SQA", "internal": "imda_part3_30s_sqa_human_test", "description": "Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 3.", "hfLink": null, "stats": { "num_rows": 100, "audio_length": { "min": 23.61, "max": 29.97, "mean": 28.29, "median": 28.79, "std": 1.44, "total_hours": 0.79, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 100, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "What does Speaker1 recount about their encounter with the person at the bus stop?", "answer": "Speaker1 recalls running after the person, apologizing, and briefly talking before the person left in an Uber. It was their last encounter, as they no longer talk or hang out.", "audioFile": "examples/imda_part3_30s_sqa_human_test/example_0.wav" }, { "instruction": "What are Speaker1 and Speaker2's views on looking for a partner or dating?", "answer": "Speaker2 mentions they are open to dating, including blind dates, but not interested in looking for a wife at the moment.", "audioFile": "examples/imda_part3_30s_sqa_human_test/example_1.wav" } ] }, { "display": "MNSC-PART4-SQA", "internal": "imda_part4_30s_sqa_human_test", "description": "Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 4.", "hfLink": null, "stats": { "num_rows": 100, "audio_length": { "min": 18.1, "max": 29.97, "mean": 27.78, "median": 28.27, "std": 2.03, "total_hours": 0.77, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 100, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "How long does Speaker1 say it initially takes to walk to their parents' house?", "answer": "Speaker1 initially claims it takes only 5 minutes or so.", "audioFile": "examples/imda_part4_30s_sqa_human_test/example_0.wav" }, { "instruction": "What kind of travel experience does Speaker2 prefer?", "answer": "Speaker2 prefers visiting places to sightsee or relax, enjoying sunsets and drinking tea, rather than coffee or busy activities.", "audioFile": "examples/imda_part4_30s_sqa_human_test/example_1.wav" } ] }, { "display": "MNSC-PART5-SQA", "internal": "imda_part5_30s_sqa_human_test", "description": "Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 5.", "hfLink": null, "stats": { "num_rows": 100, "audio_length": { "min": 21.39, "max": 29.99, "mean": 27.53, "median": 27.91, "std": 1.82, "total_hours": 0.76, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 100, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "How does Speaker1 describe their BMT experience with their platoon commander?", "answer": "Speaker1 describes their platoon commander as someone with a lot of nonsense but who made their BMT life interesting and eventful, making them feel the army wasn’t so bad.", "audioFile": "examples/imda_part5_30s_sqa_human_test/example_0.wav" }, { "instruction": "What action did Speaker1's dad take to try to resolve the issue with their internet service?", "answer": "Speaker1's dad repeatedly called them every two to three days.", "audioFile": "examples/imda_part5_30s_sqa_human_test/example_1.wav" } ] }, { "display": "MNSC-PART6-SQA", "internal": "imda_part6_30s_sqa_human_test", "description": "Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 6.", "hfLink": null, "stats": { "num_rows": 100, "audio_length": { "min": 19.24, "max": 30.0, "mean": 27.41, "median": 28.19, "std": 2.38, "total_hours": 0.76, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 100, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Are certain programs included in all school curriculums?", "answer": "Yes, programs like character education and stress coping skills are included in all school curriculums.", "audioFile": "examples/imda_part6_30s_sqa_human_test/example_0.wav" }, { "instruction": "What procedures does Speaker1 inquire about regarding new HDB flats?", "answer": "Speaker1 inquires about procedures for newly available HDB flats near Macpherson Secondary School.", "audioFile": "examples/imda_part6_30s_sqa_human_test/example_1.wav" } ] }, { "display": "SG-Streets-30 [SEA]", "internal": "sqa_sg_streets_30", "description": "SQ Streets English SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "stats": {} } ], "data": { "columns": [ { "key": "imda_part3_30s_sqa_human_test", "display": "MNSC-PART3-SQA", "description": "Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 3.", "hfLink": null, "isWer": false }, { "key": "imda_part4_30s_sqa_human_test", "display": "MNSC-PART4-SQA", "description": "Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 4.", "hfLink": null, "isWer": false }, { "key": "imda_part5_30s_sqa_human_test", "display": "MNSC-PART5-SQA", "description": "Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 5.", "hfLink": null, "isWer": false }, { "key": "imda_part6_30s_sqa_human_test", "display": "MNSC-PART6-SQA", "description": "Multitak National Speech Corpus (MNSC) dataset, Question answering task, Part 6.", "hfLink": null, "isWer": false }, { "key": "sqa_sg_streets_30", "display": "SG-Streets-30 [SEA]", "description": "SQ Streets English SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "sqa_sg_streets_30": 81.26315789, "average": 81.2632 }, { "model": "MERaLiON-3-10B", "imda_part5_30s_sqa_human_test": 75.4, "imda_part3_30s_sqa_human_test": 66.2, "imda_part6_30s_sqa_human_test": 72.6, "imda_part4_30s_sqa_human_test": 65.4, "average": 73.8989, "sqa_sg_streets_30": 89.89473684, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 70.1663, "imda_part3_30s_sqa_human_test": 59.4, "imda_part4_30s_sqa_human_test": 63.0, "imda_part5_30s_sqa_human_test": 72.0, "imda_part6_30s_sqa_human_test": 71.8, "sqa_sg_streets_30": 84.63157895 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 64.4337, "imda_part3_30s_sqa_human_test": 52.6, "imda_part4_30s_sqa_human_test": 54.6, "imda_part5_30s_sqa_human_test": 61.4, "imda_part6_30s_sqa_human_test": 70.2, "sqa_sg_streets_30": 83.36842105 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 64.1053, "imda_part3_30s_sqa_human_test": 52.4, "imda_part4_30s_sqa_human_test": 54.4, "imda_part5_30s_sqa_human_test": 66.0, "imda_part6_30s_sqa_human_test": 69.2, "sqa_sg_streets_30": 78.52631579 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 63.7095, "imda_part3_30s_sqa_human_test": 55.2, "imda_part4_30s_sqa_human_test": 50.0, "imda_part5_30s_sqa_human_test": 63.0, "imda_part6_30s_sqa_human_test": 67.4, "sqa_sg_streets_30": 82.94736842 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 61.95, "imda_part3_30s_sqa_human_test": 55.0, "imda_part4_30s_sqa_human_test": 56.4, "imda_part5_30s_sqa_human_test": 64.6, "imda_part6_30s_sqa_human_test": 71.8 }, { "model": "MERaLiON-3-3B-ASR", "imda_part3_30s_sqa_human_test": 58.4, "imda_part5_30s_sqa_human_test": 65.4, "average": 61.9, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 61.85, "imda_part3_30s_sqa_human_test": 55.2, "imda_part4_30s_sqa_human_test": 59.2, "imda_part5_30s_sqa_human_test": 64.4, "imda_part6_30s_sqa_human_test": 68.6 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 61.7516, "imda_part3_30s_sqa_human_test": 54.2, "imda_part4_30s_sqa_human_test": 52.0, "imda_part5_30s_sqa_human_test": 62.8, "imda_part6_30s_sqa_human_test": 64.6, "sqa_sg_streets_30": 75.15789474 }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 51.2, "imda_part3_30s_sqa_human_test": 45.2, "imda_part4_30s_sqa_human_test": 46.6, "imda_part5_30s_sqa_human_test": 50.8, "imda_part6_30s_sqa_human_test": 62.2 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "sqa_sg_streets_30": 69.05263158, "imda_part3_30s_sqa_human_test": 38.8, "imda_part5_30s_sqa_human_test": 45.6, "average": 48.7705, "imda_part4_30s_sqa_human_test": 44.8, "imda_part6_30s_sqa_human_test": 45.6, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 46.7, "imda_part3_30s_sqa_human_test": 42.0, "imda_part4_30s_sqa_human_test": 39.6, "imda_part5_30s_sqa_human_test": 51.6, "imda_part6_30s_sqa_human_test": 53.6 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 43.2, "imda_part3_30s_sqa_human_test": 42.0, "imda_part4_30s_sqa_human_test": 35.4, "imda_part5_30s_sqa_human_test": 45.8, "imda_part6_30s_sqa_human_test": 49.6 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 42.3, "imda_part3_30s_sqa_human_test": 32.2, "imda_part4_30s_sqa_human_test": 37.8, "imda_part5_30s_sqa_human_test": 47.8, "imda_part6_30s_sqa_human_test": 51.4 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 41.08, "imda_part3_30s_sqa_human_test": 44.2, "imda_part4_30s_sqa_human_test": 46.2, "imda_part5_30s_sqa_human_test": 54.8, "imda_part6_30s_sqa_human_test": 60.2, "sqa_sg_streets_30": 0.0 } ], "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false } }, { "key": "SDS-Singlish", "title": "Task: Spoken Dialogue Summarization - Singlish", "taskName": "sds_singlish", "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false, "datasets": [ { "display": "MNSC-PART3-SDS", "internal": "imda_part3_30s_ds_human_test", "description": "Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 3.", "hfLink": null, "stats": { "num_rows": 100, "audio_length": { "min": 23.61, "max": 29.97, "mean": 28.29, "median": 28.79, "std": 1.44, "total_hours": 0.79, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 100, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Synthesize the dialogue into a concise summary, focusing on key discussions and decisions.", "answer": "Speaker1 shares details of a brief interaction at a bus stop where they apologized to someone before the person left in an Uber. They mention it as their last encounter since they no longer stay in touch.", "audioFile": "examples/imda_part3_30s_ds_human_test/example_0.wav" }, { "instruction": "Develop a summary that succinctly captures the essential points and results of the dialogue.", "answer": "Speaker1 and Speaker2 discuss dating, with Speaker2 open to casual dates but not actively searching for a long-term partner, adding humor to the conversation.", "audioFile": "examples/imda_part3_30s_ds_human_test/example_1.wav" } ] }, { "display": "MNSC-PART4-SDS", "internal": "imda_part4_30s_ds_human_test", "description": "Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 4.", "hfLink": null, "stats": { "num_rows": 100, "audio_length": { "min": 18.1, "max": 29.97, "mean": 27.78, "median": 28.27, "std": 2.03, "total_hours": 0.77, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 100, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Compose a brief summary emphasizing the key topics and decisions within the dialogue.", "answer": "The speakers debate travel time to Speaker1's parents' house, discussing walking durations with and without factoring in traffic lights.", "audioFile": "examples/imda_part4_30s_ds_human_test/example_0.wav" }, { "instruction": "Provide a succinct summary that captures the essence and key outcomes of the dialogue.", "answer": "Speaker1 and Speaker2 discussed travel preferences, with Speaker2 expressing a desire for relaxing experiences like sightseeing and enjoying sunsets.", "audioFile": "examples/imda_part4_30s_ds_human_test/example_1.wav" } ] }, { "display": "MNSC-PART5-SDS", "internal": "imda_part5_30s_ds_human_test", "description": "Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 5.", "hfLink": null, "stats": { "num_rows": 100, "audio_length": { "min": 21.39, "max": 29.99, "mean": 27.53, "median": 27.91, "std": 1.82, "total_hours": 0.76, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 100, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Craft a summary that focuses on the pivotal points and conclusions of the dialogue.", "answer": "Speaker1 and Speaker2 discuss Speaker1's BMT experience. Speaker1 shares how their platoon commander’s unconventional behavior made the training period more engaging and less daunting, shifting their perception of the army positively.", "audioFile": "examples/imda_part5_30s_ds_human_test/example_0.wav" }, { "instruction": "Forge a succinct narrative summarizing the dialogue’s main points and resolutions.", "answer": "Speaker1 discussed their difficult experience with a technical issue that didn't get resolved despite multiple attempts. Their dad had to call the company repeatedly, and eventually, they sent technicians to conduct tests and replaced the modems and routers.", "audioFile": "examples/imda_part5_30s_ds_human_test/example_1.wav" } ] }, { "display": "MNSC-PART6-SDS", "internal": "imda_part6_30s_ds_human_test", "description": "Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 6.", "hfLink": null, "stats": { "num_rows": 100, "audio_length": { "min": 19.24, "max": 30.0, "mean": 27.41, "median": 28.19, "std": 2.38, "total_hours": 0.76, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 100, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Forge a succinct narrative summarizing the dialogue’s main points and resolutions.", "answer": "Speaker2 explains that programs like character education are integrated into the curriculums of all schools.", "audioFile": "examples/imda_part6_30s_ds_human_test/example_0.wav" }, { "instruction": "Generate a summary that distills the dialogue into its most important discussions and decisions.", "answer": "Speaker1 expresses interest in HDB flats near Macpherson and provides their full name and contact details for the inquiry.", "audioFile": "examples/imda_part6_30s_ds_human_test/example_1.wav" } ] } ], "data": { "columns": [ { "key": "imda_part3_30s_ds_human_test", "display": "MNSC-PART3-SDS", "description": "Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 3.", "hfLink": null, "isWer": false }, { "key": "imda_part4_30s_ds_human_test", "display": "MNSC-PART4-SDS", "description": "Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 4.", "hfLink": null, "isWer": false }, { "key": "imda_part5_30s_ds_human_test", "display": "MNSC-PART5-SDS", "description": "Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 5.", "hfLink": null, "isWer": false }, { "key": "imda_part6_30s_ds_human_test", "display": "MNSC-PART6-SDS", "description": "Multitak National Speech Corpus (MNSC) dataset, dialogue summarization task, Part 6.", "hfLink": null, "isWer": false } ], "rows": [ { "model": "MERaLiON-3-10B", "imda_part6_30s_ds_human_test": 64.0, "imda_part3_30s_ds_human_test": 59.8, "imda_part4_30s_ds_human_test": 59.0, "imda_part5_30s_ds_human_test": 65.0, "average": 61.95, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-3-3B-ASR", "imda_part4_30s_ds_human_test": 51.8, "imda_part6_30s_ds_human_test": 61.4, "average": 56.6, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 55.8, "imda_part3_30s_ds_human_test": 53.0, "imda_part4_30s_ds_human_test": 49.6, "imda_part5_30s_ds_human_test": 58.2, "imda_part6_30s_ds_human_test": 62.4 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 53.6, "imda_part3_30s_ds_human_test": 47.8, "imda_part4_30s_ds_human_test": 46.4, "imda_part5_30s_ds_human_test": 54.6, "imda_part6_30s_ds_human_test": 65.6 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 53.1, "imda_part3_30s_ds_human_test": 49.8, "imda_part4_30s_ds_human_test": 46.6, "imda_part5_30s_ds_human_test": 55.4, "imda_part6_30s_ds_human_test": 60.6 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 50.75, "imda_part3_30s_ds_human_test": 43.6, "imda_part4_30s_ds_human_test": 42.8, "imda_part5_30s_ds_human_test": 55.6, "imda_part6_30s_ds_human_test": 61.0 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 48.55, "imda_part3_30s_ds_human_test": 42.2, "imda_part4_30s_ds_human_test": 40.2, "imda_part5_30s_ds_human_test": 51.8, "imda_part6_30s_ds_human_test": 60.0 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 46.75, "imda_part3_30s_ds_human_test": 42.8, "imda_part4_30s_ds_human_test": 33.2, "imda_part5_30s_ds_human_test": 52.2, "imda_part6_30s_ds_human_test": 58.8 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 45.45, "imda_part3_30s_ds_human_test": 42.0, "imda_part4_30s_ds_human_test": 35.2, "imda_part5_30s_ds_human_test": 49.6, "imda_part6_30s_ds_human_test": 55.0 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 43.15, "imda_part3_30s_ds_human_test": 39.8, "imda_part4_30s_ds_human_test": 31.6, "imda_part5_30s_ds_human_test": 42.8, "imda_part6_30s_ds_human_test": 58.4 }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 39.45, "imda_part3_30s_ds_human_test": 31.6, "imda_part4_30s_ds_human_test": 31.6, "imda_part5_30s_ds_human_test": 45.2, "imda_part6_30s_ds_human_test": 49.4 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "imda_part5_30s_ds_human_test": 46.0, "imda_part6_30s_ds_human_test": 49.6, "imda_part4_30s_ds_human_test": 32.8, "imda_part3_30s_ds_human_test": 27.6, "average": 39.0, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 36.3, "imda_part3_30s_ds_human_test": 33.8, "imda_part4_30s_ds_human_test": 24.8, "imda_part5_30s_ds_human_test": 40.4, "imda_part6_30s_ds_human_test": 46.2 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 25.25, "imda_part3_30s_ds_human_test": 16.4, "imda_part4_30s_ds_human_test": 16.0, "imda_part5_30s_ds_human_test": 28.2, "imda_part6_30s_ds_human_test": 40.4 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 14.4, "imda_part3_30s_ds_human_test": 9.0, "imda_part4_30s_ds_human_test": 7.4, "imda_part5_30s_ds_human_test": 16.0, "imda_part6_30s_ds_human_test": 25.2 } ], "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false } }, { "key": "Speech Instruction", "title": "Task: Speech Instruction", "taskName": "speech_instruction", "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false, "datasets": [ { "display": "OpenHermes-Audio", "internal": "openhermes_audio_test", "description": "Test set for spoken instructions. Synthesized from the OpenHermes dataset.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/openhermes_instruction_test", "stats": { "num_rows": 100, "audio_length": { "min": 2.04, "max": 15.78, "mean": 5.95, "median": 5.29, "std": 2.67, "total_hours": 0.17, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 100, 0, 0, 0, 0 ] } } } }, { "display": "ALPACA-Audio", "internal": "alpaca_audio_test", "description": "Spoken version of the ALPACA dataset, used for evaluating instruction following in audio.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/alpaca_audio_test", "stats": { "num_rows": 100, "audio_length": { "min": 1.8, "max": 8.85, "mean": 4.32, "median": 4.13, "std": 1.36, "total_hours": 0.12, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 100, 0, 0, 0, 0 ] } } } } ], "data": { "columns": [ { "key": "openhermes_audio_test", "display": "OpenHermes-Audio", "description": "Test set for spoken instructions. Synthesized from the OpenHermes dataset.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/openhermes_instruction_test", "isWer": false }, { "key": "alpaca_audio_test", "display": "ALPACA-Audio", "description": "Spoken version of the ALPACA dataset, used for evaluating instruction following in audio.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/alpaca_audio_test", "isWer": false } ], "rows": [ { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 70.8, "openhermes_audio_test": 66.4, "alpaca_audio_test": 75.2 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 70.2, "openhermes_audio_test": 66.2, "alpaca_audio_test": 74.2 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 66.9, "openhermes_audio_test": 66.2, "alpaca_audio_test": 67.6 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 65.0, "openhermes_audio_test": 66.0, "alpaca_audio_test": 64.0 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "openhermes_audio_test": 60.0, "alpaca_audio_test": 64.2, "average": 62.1, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 58.3, "openhermes_audio_test": 57.4, "alpaca_audio_test": 59.2 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 48.7, "openhermes_audio_test": 44.8, "alpaca_audio_test": 52.6 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 36.2, "openhermes_audio_test": 39.0, "alpaca_audio_test": 33.4 }, { "model": "MERaLiON-3-3B-ASR", "openhermes_audio_test": 20.6, "average": 20.6, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 20.4, "openhermes_audio_test": 19.2, "alpaca_audio_test": 21.6 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 19.1, "openhermes_audio_test": 12.6, "alpaca_audio_test": 25.6 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 13.4, "openhermes_audio_test": 10.0, "alpaca_audio_test": 16.8 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 12.9, "openhermes_audio_test": 15.4, "alpaca_audio_test": 10.4 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 10.2, "openhermes_audio_test": 10.6, "alpaca_audio_test": 9.8 } ], "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false } }, { "key": "Audio Captioning", "title": "Task: Audio Captioning", "taskName": "audio_captioning", "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false, "datasets": [ { "display": "WavCaps", "internal": "wavcaps_test", "description": "WavCaps is a dataset for testing audio captioning, where models generate textual descriptions of audio clips.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/wavcaps_test", "stats": { "num_rows": 1730, "audio_length": { "min": 1.0, "max": 30.97, "mean": 10.23, "median": 10.0, "std": 6.48, "total_hours": 4.92, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 1707, 23, 0, 0, 0 ] } } } }, { "display": "AudioCaps", "internal": "audiocaps_test", "description": "AudioCaps dataset, used for generating captions from general audio events.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/audiocaps_test", "stats": { "num_rows": 4400, "audio_length": { "min": 1.74, "max": 10.0, "mean": 9.86, "median": 10.0, "std": 0.66, "total_hours": 12.05, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 4400, 0, 0, 0, 0 ] } } } } ], "data": { "columns": [ { "key": "audiocaps_test", "display": "AudioCaps", "description": "AudioCaps dataset, used for generating captions from general audio events.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/audiocaps_test", "isWer": false }, { "key": "wavcaps_test", "display": "WavCaps", "description": "WavCaps is a dataset for testing audio captioning, where models generate textual descriptions of audio clips.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/wavcaps_test", "isWer": false } ], "rows": [ { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 45.0895, "audiocaps_test": 51.959, "wavcaps_test": 38.22 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 39.9885, "audiocaps_test": 47.041, "wavcaps_test": 32.936 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 39.2, "audiocaps_test": 43.695, "wavcaps_test": 34.705 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 37.2785, "audiocaps_test": 40.777, "wavcaps_test": 33.78 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 36.976, "audiocaps_test": 39.386, "wavcaps_test": 34.566 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 35.6045, "audiocaps_test": 36.041, "wavcaps_test": 35.168 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 34.467, "audiocaps_test": 35.373, "wavcaps_test": 33.561 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 33.2435, "audiocaps_test": 35.077, "wavcaps_test": 31.41 }, { "model": "MERaLiON-3-10B", "audiocaps_test": 34.5455, "wavcaps_test": 30.9364, "average": 32.7409, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-3-3B-ASR", "audiocaps_test": 32.1409, "average": 32.1409, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 31.896, "audiocaps_test": 37.7, "wavcaps_test": 26.092 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 30.832, "audiocaps_test": 33.595, "wavcaps_test": 28.069 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 28.8805, "audiocaps_test": 35.241, "wavcaps_test": 22.52 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "audiocaps_test": 11.0909, "wavcaps_test": 18.1387, "average": 14.6148, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 6.201, "audiocaps_test": 5.5, "wavcaps_test": 6.902 } ], "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false } }, { "key": "Audio-Scene QA", "title": "Task: Audio Scene Question Answering", "taskName": "audio_scene_question_answering", "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false, "datasets": [ { "display": "Clotho-AQA", "internal": "clotho_aqa_test", "description": "Clotho dataset adapted for audio-based question answering, containing audio clips and questions.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/clotho_asqa_test_v2", "stats": { "num_rows": 2057, "audio_length": { "min": 15.03, "max": 29.98, "mean": 22.56, "median": 22.42, "std": 4.29, "total_hours": 12.89, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 2057, 0, 0, 0, 0 ] } } } }, { "display": "WavCaps-QA", "internal": "wavcaps_qa_test", "description": "Question-answering test dataset derived from WavCaps, focusing on audio content.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/wavcaps_qa_test_v3", "stats": { "num_rows": 304, "audio_length": { "min": 1.0, "max": 30.63, "mean": 10.29, "median": 10.0, "std": 6.33, "total_hours": 0.87, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 301, 3, 0, 0, 0 ] } } } }, { "display": "AudioCaps-QA", "internal": "audiocaps_qa_test", "description": "AudioCaps adapted for question-answering tasks, using audio events as input for Q&A.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/audiocaps_qa_test_v3", "stats": { "num_rows": 313, "audio_length": { "min": 3.27, "max": 10.0, "mean": 9.86, "median": 10.0, "std": 0.71, "total_hours": 0.86, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 313, 0, 0, 0, 0 ] } } } } ], "data": { "columns": [ { "key": "clotho_aqa_test", "display": "Clotho-AQA", "description": "Clotho dataset adapted for audio-based question answering, containing audio clips and questions.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/clotho_asqa_test_v2", "isWer": false }, { "key": "audiocaps_qa_test", "display": "AudioCaps-QA", "description": "AudioCaps adapted for question-answering tasks, using audio events as input for Q&A.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/audiocaps_qa_test_v3", "isWer": false }, { "key": "wavcaps_qa_test", "display": "WavCaps-QA", "description": "Question-answering test dataset derived from WavCaps, focusing on audio content.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/wavcaps_qa_test_v3", "isWer": false } ], "rows": [ { "model": "MERaLiON-3-10B", "audiocaps_qa_test": 52.0767, "average": 52.9139, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B", "wavcaps_qa_test": 48.0263, "clotho_aqa_test": 58.6388 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 52.208, "clotho_aqa_test": 62.674, "audiocaps_qa_test": 48.818, "wavcaps_qa_test": 45.132 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 51.817, "clotho_aqa_test": 58.192, "audiocaps_qa_test": 50.351, "wavcaps_qa_test": 46.908 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 51.6187, "clotho_aqa_test": 61.935, "audiocaps_qa_test": 50.224, "wavcaps_qa_test": 42.697 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 51.14, "clotho_aqa_test": 58.201, "audiocaps_qa_test": 50.351, "wavcaps_qa_test": 44.868 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 50.1937, "clotho_aqa_test": 53.894, "audiocaps_qa_test": 54.121, "wavcaps_qa_test": 42.566 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 49.512, "clotho_aqa_test": 56.5, "audiocaps_qa_test": 49.01, "wavcaps_qa_test": 43.026 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 48.123, "clotho_aqa_test": 52.649, "audiocaps_qa_test": 48.562, "wavcaps_qa_test": 43.158 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 47.0483, "clotho_aqa_test": 50.92, "audiocaps_qa_test": 45.751, "wavcaps_qa_test": 44.474 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 46.1413, "clotho_aqa_test": 50.54, "audiocaps_qa_test": 44.792, "wavcaps_qa_test": 43.092 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 45.669, "clotho_aqa_test": 46.592, "audiocaps_qa_test": 50.415, "wavcaps_qa_test": 40.0 }, { "model": "MERaLiON-3-3B-ASR", "audiocaps_qa_test": 43.6422, "average": 43.6422, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 42.217, "clotho_aqa_test": 48.371, "audiocaps_qa_test": 40.319, "wavcaps_qa_test": 37.961 }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 33.034, "clotho_aqa_test": 43.012, "audiocaps_qa_test": 29.84, "wavcaps_qa_test": 26.25 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "clotho_aqa_test": 23.053, "average": 21.653, "audiocaps_qa_test": 20.0639, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT", "wavcaps_qa_test": 21.8421 } ], "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false } }, { "key": "Accent Recognition", "title": "Task: Accent Recognition", "taskName": "accent_recognition", "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false, "datasets": [ { "display": "VoxCeleb-Accent", "internal": "voxceleb_accent_test", "description": "Test dataset for accent recognition, based on VoxCeleb, a large speaker identification dataset.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/voxceleb_accent_test", "stats": { "num_rows": 4874, "audio_length": { "min": 3.96, "max": 69.04, "mean": 8.28, "median": 6.48, "std": 5.74, "total_hours": 11.2, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 4818, 56, 0, 0, 0 ] } } } }, { "display": "MNSC-AR-Sentence", "internal": "imda_ar_sentence", "description": "Accent recognition based on the IMDA NSC dataset, focusing on sentence-level accents.", "hfLink": null, "stats": { "num_rows": 6000, "audio_length": { "min": 1.86, "max": 14.85, "mean": 5.4, "median": 5.03, "std": 1.98, "total_hours": 8.99, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 6000, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Can you identify if the speaker's accent reflects any Singaporean tones in their English?", "answer": "The speaker's accent suggests they are a Malay-Speaking community of Singapore.", "audioFile": "examples/imda_ar_sentence/example_0.wav" }, { "instruction": "Can you infer the speaker's accent from this audio clip?", "answer": "The speaker is likely from Singapore's Malay-Speaking community, whose first language is Malay.", "audioFile": "examples/imda_ar_sentence/example_1.wav" } ] }, { "display": "MNSC-AR-Dialogue", "internal": "imda_ar_dialogue", "description": "Accent recognition based on the IMDA NSC dataset, focusing on dialogue-level accents.", "hfLink": null, "stats": { "num_rows": 3000, "audio_length": { "min": 15.02, "max": 30.0, "mean": 26.29, "median": 27.31, "std": 3.36, "total_hours": 21.91, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 3000, 0, 0, 0, 0 ] } } }, "examples": [ { "instruction": "Can you describe the accent of the speakers?", "answer": "Both speakers have a Malay accent. They are likely from the Malay-Speaking community of Singapore and speak with a Singaporean accent.", "audioFile": "examples/imda_ar_dialogue/example_0.wav" }, { "instruction": "Can you describe the accent of the speakers?", "answer": "The first speaker has a Malay accent, likely from the Malay-Speaking community of Singapore, while the second speaker speaks good English with a Singapore accent.", "audioFile": "examples/imda_ar_dialogue/example_1.wav" } ] } ], "data": { "columns": [ { "key": "voxceleb_accent_test", "display": "VoxCeleb-Accent", "description": "Test dataset for accent recognition, based on VoxCeleb, a large speaker identification dataset.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/voxceleb_accent_test", "isWer": false }, { "key": "imda_ar_sentence", "display": "MNSC-AR-Sentence", "description": "Accent recognition based on the IMDA NSC dataset, focusing on sentence-level accents.", "hfLink": null, "isWer": false }, { "key": "imda_ar_dialogue", "display": "MNSC-AR-Dialogue", "description": "Accent recognition based on the IMDA NSC dataset, focusing on dialogue-level accents.", "hfLink": null, "isWer": false } ], "rows": [ { "model": "MERaLiON-3-10B", "voxceleb_accent_test": 65.2749, "imda_ar_dialogue": 82.6333, "average": 79.1958, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B", "imda_ar_sentence": 89.6792 }, { "model": "MERaLiON-3-3B-ASR", "imda_ar_dialogue": 56.6167, "voxceleb_accent_test": 77.1748, "average": 66.8958, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 60.0547, "voxceleb_accent_test": 66.598, "imda_ar_sentence": 59.733, "imda_ar_dialogue": 53.833 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 47.7887, "voxceleb_accent_test": 18.383, "imda_ar_sentence": 58.25, "imda_ar_dialogue": 66.733 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 43.7997, "voxceleb_accent_test": 47.066, "imda_ar_sentence": 6.333, "imda_ar_dialogue": 78.0 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 41.8153, "voxceleb_accent_test": 40.788, "imda_ar_sentence": 30.325, "imda_ar_dialogue": 54.333 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "voxceleb_accent_test": 61.2331, "average": 27.0833, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT", "imda_ar_sentence": 5.2167, "imda_ar_dialogue": 14.8 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 17.5503, "voxceleb_accent_test": 48.051, "imda_ar_sentence": 3.933, "imda_ar_dialogue": 0.667 }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 14.2943, "voxceleb_accent_test": 39.967, "imda_ar_sentence": 2.683, "imda_ar_dialogue": 0.233 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 11.5773, "voxceleb_accent_test": 31.699, "imda_ar_sentence": 2.833, "imda_ar_dialogue": 0.2 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 10.9017, "voxceleb_accent_test": 29.188, "imda_ar_sentence": 2.55, "imda_ar_dialogue": 0.967 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 10.1437, "voxceleb_accent_test": 9.848, "imda_ar_sentence": 3.85, "imda_ar_dialogue": 16.733 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 3.0973, "voxceleb_accent_test": 2.626, "imda_ar_sentence": 6.133, "imda_ar_dialogue": 0.533 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 0.5873, "voxceleb_accent_test": 1.662, "imda_ar_sentence": 0.067, "imda_ar_dialogue": 0.033 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 0.4787, "voxceleb_accent_test": 0.903, "imda_ar_sentence": 0.1, "imda_ar_dialogue": 0.433 } ], "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false } }, { "key": "Gender Recognition", "title": "Task: Gender Recognition", "taskName": "gender_recognition", "metric": "accuracy / judge", "metricInfo": "Scale from 0-100. The higher, the better. AudioBench columns are Model-as-a-Judge (LLaMA-3-70B); AudioBench-SEA columns are accuracy.", "ascending": false, "datasets": [ { "display": "VoxCeleb-Gender", "internal": "voxceleb_gender_test", "description": "Test dataset for gender classification, also derived from VoxCeleb.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/voxceleb_gender_test", "stats": { "num_rows": 4874, "audio_length": { "min": 3.96, "max": 69.04, "mean": 8.28, "median": 6.48, "std": 5.74, "total_hours": 11.2, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 4818, 56, 0, 0, 0 ] } } } }, { "display": "IEMOCAP-Gender", "internal": "iemocap_gender_test", "description": "Gender classification based on the IEMOCAP dataset.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/iemocap_gender_recognition", "stats": { "num_rows": 1004, "audio_length": { "min": 0.84, "max": 34.14, "mean": 4.42, "median": 3.42, "std": 3.26, "total_hours": 1.23, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 1003, 1, 0, 0, 0 ] } } } }, { "display": "GR-Cv21-ID-30 [SEA]", "internal": "gr_cv21_id_30", "description": "Gender CV21 Indonesian 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Cv21-TA-30 [SEA]", "internal": "gr_cv21_ta_30", "description": "Gender CV21 Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Cv21-TH-30 [SEA]", "internal": "gr_cv21_th_30", "description": "Gender CV21 Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Cv21-VI-30 [SEA]", "internal": "gr_cv21_vi_30", "description": "Gender CV21 Vietnamese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Cv21-ZH-30 [SEA]", "internal": "gr_cv21_zh_30", "description": "Gender CV21 Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Emota-TA-30 [SEA]", "internal": "gr_emota_ta_30", "description": "Gender EMOTA Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Fleurs-EN-30 [SEA]", "internal": "gr_fleurs_en_30", "description": "Gender Fleurs English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Fleurs-KM-30 [SEA]", "internal": "gr_fleurs_km_30", "description": "Gender Fleurs Khmer 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Indowave-ID-30 [SEA]", "internal": "gr_indowave_id_30", "description": "Gender InDowave Indonesian 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-M3ed-30 [SEA]", "internal": "gr_m3ed_30", "description": "Gender M3ED English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Openslr-TA-30 [SEA]", "internal": "gr_openslr_ta_30", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-SG-Streets-Utterance-30 [SEA]", "internal": "gr_sg_streets_utterance_30", "description": "Gender SG Streets Utterance 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Smaldusc-30 [SEA]", "internal": "gr_smaldusc_30", "description": "Gender SMALDUSC 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-THAI-Elderly-TH-30 [SEA]", "internal": "gr_thai_elderly_th_30", "description": "Gender Thai Elderly Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-THAI-SER-TH-30 [SEA]", "internal": "gr_thai_ser_th_30", "description": "Gender Thai SER Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Vietnam-Celeb-30 [SEA]", "internal": "gr_vietnam_celeb_30", "description": "Gender Vietnam Celeb 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Cv21-EN-30 [SEA]", "internal": "gr_cv21_en_30", "description": "Gender CV21 English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/MERaLiON/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "GR-Sfdusc-30 [SEA]", "internal": "gr_sfdusc_30", "description": "Gender SFDUSC Tagalog 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/MERaLiON/sea_audiobench_datasets_PQA", "stats": {} } ], "data": { "columns": [ { "key": "voxceleb_gender_test", "display": "VoxCeleb-Gender", "description": "Test dataset for gender classification, also derived from VoxCeleb.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/voxceleb_gender_test", "isWer": false }, { "key": "iemocap_gender_test", "display": "IEMOCAP-Gender", "description": "Gender classification based on the IEMOCAP dataset.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/iemocap_gender_recognition", "isWer": false }, { "key": "gr_cv21_id_30", "display": "GR-Cv21-ID-30 [SEA]", "description": "Gender CV21 Indonesian 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_cv21_ta_30", "display": "GR-Cv21-TA-30 [SEA]", "description": "Gender CV21 Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_cv21_th_30", "display": "GR-Cv21-TH-30 [SEA]", "description": "Gender CV21 Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_cv21_vi_30", "display": "GR-Cv21-VI-30 [SEA]", "description": "Gender CV21 Vietnamese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_cv21_zh_30", "display": "GR-Cv21-ZH-30 [SEA]", "description": "Gender CV21 Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_emota_ta_30", "display": "GR-Emota-TA-30 [SEA]", "description": "Gender EMOTA Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_fleurs_en_30", "display": "GR-Fleurs-EN-30 [SEA]", "description": "Gender Fleurs English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_fleurs_km_30", "display": "GR-Fleurs-KM-30 [SEA]", "description": "Gender Fleurs Khmer 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_indowave_id_30", "display": "GR-Indowave-ID-30 [SEA]", "description": "Gender InDowave Indonesian 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_m3ed_30", "display": "GR-M3ed-30 [SEA]", "description": "Gender M3ED English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_openslr_ta_30", "display": "GR-Openslr-TA-30 [SEA]", "description": "AudioBench-SEA v2 dataset. (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_sg_streets_utterance_30", "display": "GR-SG-Streets-Utterance-30 [SEA]", "description": "Gender SG Streets Utterance 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_smaldusc_30", "display": "GR-Smaldusc-30 [SEA]", "description": "Gender SMALDUSC 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_thai_elderly_th_30", "display": "GR-THAI-Elderly-TH-30 [SEA]", "description": "Gender Thai Elderly Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_thai_ser_th_30", "display": "GR-THAI-SER-TH-30 [SEA]", "description": "Gender Thai SER Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_vietnam_celeb_30", "display": "GR-Vietnam-Celeb-30 [SEA]", "description": "Gender Vietnam Celeb 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_cv21_en_30", "display": "GR-Cv21-EN-30 [SEA]", "description": "Gender CV21 English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/MERaLiON/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "gr_sfdusc_30", "display": "GR-Sfdusc-30 [SEA]", "description": "Gender SFDUSC Tagalog 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/MERaLiON/sea_audiobench_datasets_PQA", "isWer": false } ], "rows": [ { "model": "MERaLiON-3-10B", "gr_cv21_id_30": 96.9, "gr_cv21_ta_30": 96.8, "gr_cv21_th_30": 97.59036145, "gr_cv21_vi_30": 98.69281046, "gr_cv21_zh_30": 97.9, "gr_emota_ta_30": 99.89316239, "gr_fleurs_en_30": 99.38176198, "gr_fleurs_km_30": 100.0, "gr_indowave_id_30": 99.33333333, "gr_m3ed_30": 92.7, "gr_openslr_ta_30": 100.0, "gr_sg_streets_utterance_30": 99.59349593, "gr_smaldusc_30": 99.2, "gr_thai_elderly_th_30": 99.19354839, "gr_thai_ser_th_30": 90.78534031, "gr_vietnam_celeb_30": 73.6, "average": 96.4339, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B", "iemocap_gender_test": 95.71713147, "voxceleb_gender_test": 99.49733279, "gr_cv21_en_30": 92.0, "gr_sfdusc_30": 99.9 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "gr_cv21_id_30": 97.2, "gr_cv21_ta_30": 97.0, "gr_cv21_th_30": 97.1888, "gr_cv21_vi_30": 98.8235, "gr_cv21_zh_30": 98.2, "gr_emota_ta_30": 99.8932, "gr_fleurs_en_30": 100.0, "gr_fleurs_km_30": 100.0, "gr_indowave_id_30": 100.0, "gr_m3ed_30": 94.9, "gr_openslr_ta_30": 99.1, "gr_sg_streets_utterance_30": 100.0, "gr_smaldusc_30": 98.6, "gr_thai_elderly_th_30": 99.0927, "gr_thai_ser_th_30": 90.8901, "gr_vietnam_celeb_30": 73.8, "average": 96.0993, "gr_cv21_en_30": 93.3, "gr_sfdusc_30": 91.8 }, { "model": "MERaLiON-GI-v1", "gr_cv21_id_30": 97.1, "gr_cv21_ta_30": 94.2, "gr_cv21_th_30": 97.72423025, "gr_cv21_vi_30": 99.08496732, "gr_cv21_zh_30": 98.0, "gr_emota_ta_30": 99.14529915, "gr_fleurs_en_30": 100.0, "gr_fleurs_km_30": 99.60784314, "gr_indowave_id_30": 98.0, "gr_m3ed_30": 93.3, "gr_openslr_ta_30": 99.9, "gr_sg_streets_utterance_30": 99.79674797, "gr_smaldusc_30": 92.1, "gr_thai_elderly_th_30": 99.89919355, "gr_thai_ser_th_30": 86.91099476, "gr_vietnam_celeb_30": 73.4, "iemocap_gender_test": 98.80478088, "voxceleb_gender_test": 99.75379565, "average": 95.7864, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-GI-v1", "gr_cv21_en_30": 94.0, "gr_sfdusc_30": 95.0 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 91.6135, "voxceleb_gender_test": 99.118, "iemocap_gender_test": 92.809, "gr_fleurs_km_30": 78.5620915, "gr_fleurs_en_30": 99.22720247, "gr_cv21_id_30": 94.9, "gr_cv21_ta_30": 96.8, "gr_cv21_th_30": 96.78714859, "gr_cv21_vi_30": 94.77124183, "gr_cv21_zh_30": 98.3, "gr_emota_ta_30": 98.71794872, "gr_indowave_id_30": 98.0, "gr_m3ed_30": 85.7, "gr_openslr_ta_30": 99.1, "gr_sg_streets_utterance_30": 85.16260163, "gr_smaldusc_30": 93.6, "gr_thai_elderly_th_30": 97.47983871, "gr_thai_ser_th_30": 87.43455497, "gr_vietnam_celeb_30": 69.2, "gr_cv21_en_30": 91.7, "gr_sfdusc_30": 74.9 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 80.3979, "voxceleb_gender_test": 99.733, "iemocap_gender_test": 94.622, "gr_cv21_id_30": 75.3, "gr_cv21_ta_30": 69.6, "gr_cv21_th_30": 72.28915663, "gr_cv21_vi_30": 60.91503268, "gr_cv21_zh_30": 70.3, "gr_emota_ta_30": 94.76495726, "gr_fleurs_en_30": 51.77743431, "gr_fleurs_km_30": 91.50326797, "gr_indowave_id_30": 96.33333333, "gr_m3ed_30": 92.9, "gr_openslr_ta_30": 75.5, "gr_sg_streets_utterance_30": 95.52845528, "gr_smaldusc_30": 73.9, "gr_thai_elderly_th_30": 86.29032258, "gr_thai_ser_th_30": 88.90052356, "gr_vietnam_celeb_30": 72.8, "gr_cv21_en_30": 56.4, "gr_sfdusc_30": 88.6 }, { "model": "MERaLiON-3-3B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR", "gr_cv21_id_30": 71.2, "gr_cv21_th_30": 78.31325301, "gr_cv21_vi_30": 74.24836601, "gr_cv21_zh_30": 62.4, "gr_fleurs_en_30": 87.63523957, "gr_fleurs_km_30": 65.49019608, "gr_indowave_id_30": 82.33333333, "gr_m3ed_30": 66.2, "gr_sg_streets_utterance_30": 85.16260163, "gr_smaldusc_30": 65.8, "gr_thai_elderly_th_30": 77.16733871, "gr_vietnam_celeb_30": 63.3, "average": 75.6124, "gr_emota_ta_30": 66.72008547, "gr_openslr_ta_30": 79.5, "gr_thai_ser_th_30": 84.60732984, "iemocap_gender_test": 73.65537849, "voxceleb_gender_test": 99.41526467, "gr_cv21_en_30": 77.0, "gr_sfdusc_30": 69.8, "gr_cv21_ta_30": 82.3 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 70.2353, "voxceleb_gender_test": 99.261, "iemocap_gender_test": 95.179, "gr_cv21_id_30": 56.0, "gr_cv21_ta_30": 56.3, "gr_cv21_th_30": 65.19410977, "gr_cv21_vi_30": 24.70588235, "gr_cv21_zh_30": 53.2, "gr_emota_ta_30": 81.3034188, "gr_fleurs_en_30": 46.36785162, "gr_fleurs_km_30": 76.60130719, "gr_indowave_id_30": 89.66666667, "gr_m3ed_30": 90.9, "gr_openslr_ta_30": 52.9, "gr_sg_streets_utterance_30": 94.30894309, "gr_smaldusc_30": 67.1, "gr_thai_elderly_th_30": 71.77419355, "gr_thai_ser_th_30": 87.64397906, "gr_vietnam_celeb_30": 71.3, "gr_cv21_en_30": 56.2, "gr_sfdusc_30": 68.8 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 68.1737, "voxceleb_gender_test": 88.531, "iemocap_gender_test": 80.199, "gr_cv21_id_30": 67.9, "gr_cv21_ta_30": 56.9, "gr_cv21_th_30": 61.98125837, "gr_cv21_vi_30": 86.66666667, "gr_cv21_zh_30": 70.8, "gr_emota_ta_30": 72.64957265, "gr_fleurs_en_30": 66.46058733, "gr_fleurs_km_30": 73.85620915, "gr_indowave_id_30": 72.0, "gr_m3ed_30": 71.0, "gr_openslr_ta_30": 57.1, "gr_sg_streets_utterance_30": 68.29268293, "gr_smaldusc_30": 66.8, "gr_thai_elderly_th_30": 40.22177419, "gr_thai_ser_th_30": 75.91623037, "gr_vietnam_celeb_30": 60.3, "gr_cv21_en_30": 56.8, "gr_sfdusc_30": 69.1 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 61.8229, "voxceleb_gender_test": 97.251, "iemocap_gender_test": 92.968, "gr_cv21_id_30": 43.9, "gr_cv21_ta_30": 55.1, "gr_cv21_th_30": 48.8621, "gr_cv21_vi_30": 24.183, "gr_cv21_zh_30": 53.5, "gr_emota_ta_30": 66.8803, "gr_fleurs_en_30": 57.4961, "gr_fleurs_km_30": 58.0392, "gr_indowave_id_30": 73.0, "gr_m3ed_30": 83.7, "gr_openslr_ta_30": 55.2, "gr_sg_streets_utterance_30": 89.0244, "gr_smaldusc_30": 52.0, "gr_thai_elderly_th_30": 66.2298, "gr_thai_ser_th_30": 63.8743, "gr_vietnam_celeb_30": 66.6, "gr_cv21_en_30": 44.95, "gr_sfdusc_30": 43.7 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 56.9003, "voxceleb_gender_test": 82.273, "iemocap_gender_test": 68.625, "gr_cv21_id_30": 54.8, "gr_cv21_ta_30": 50.0, "gr_cv21_th_30": 56.89424364, "gr_cv21_vi_30": 82.74509804, "gr_cv21_zh_30": 68.9, "gr_emota_ta_30": 52.77777778, "gr_fleurs_en_30": 39.25811437, "gr_fleurs_km_30": 71.50326797, "gr_indowave_id_30": 50.0, "gr_m3ed_30": 51.1, "gr_openslr_ta_30": 50.2, "gr_sg_streets_utterance_30": 62.39837398, "gr_smaldusc_30": 51.8, "gr_thai_elderly_th_30": 34.375, "gr_thai_ser_th_30": 50.15706806, "gr_vietnam_celeb_30": 50.8, "gr_cv21_en_30": 59.4, "gr_sfdusc_30": 50.0 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 56.5974, "voxceleb_gender_test": 54.083, "iemocap_gender_test": 43.367, "gr_cv21_id_30": 61.75, "gr_cv21_ta_30": 68.55, "gr_cv21_th_30": 69.61178046, "gr_cv21_vi_30": 56.73202614, "gr_cv21_zh_30": 83.5, "gr_emota_ta_30": 66.77350427, "gr_fleurs_en_30": 37.2488408, "gr_fleurs_km_30": 63.26797386, "gr_indowave_id_30": 62.66666667, "gr_m3ed_30": 47.4, "gr_openslr_ta_30": 52.8, "gr_sg_streets_utterance_30": 7.11382114, "gr_smaldusc_30": 64.1, "gr_thai_elderly_th_30": 67.23790323, "gr_thai_ser_th_30": 63.2460733, "gr_vietnam_celeb_30": 56.1, "gr_cv21_en_30": 45.8, "gr_sfdusc_30": 60.6 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 54.6446, "voxceleb_gender_test": 70.599, "iemocap_gender_test": 50.1, "gr_cv21_id_30": 54.5, "gr_cv21_ta_30": 50.0, "gr_cv21_th_30": 56.89424364, "gr_cv21_vi_30": 83.1372549, "gr_cv21_zh_30": 70.8, "gr_emota_ta_30": 52.88461538, "gr_fleurs_en_30": 36.63060278, "gr_fleurs_km_30": 72.15686275, "gr_indowave_id_30": 50.0, "gr_m3ed_30": 49.9, "gr_openslr_ta_30": 50.1, "gr_sg_streets_utterance_30": 62.19512195, "gr_smaldusc_30": 50.3, "gr_thai_elderly_th_30": 31.75403226, "gr_thai_ser_th_30": 45.34031414, "gr_vietnam_celeb_30": 49.9, "gr_cv21_en_30": 55.7, "gr_sfdusc_30": 50.0 }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 54.2405, "voxceleb_gender_test": 69.614, "iemocap_gender_test": 51.932, "gr_cv21_id_30": 54.6, "gr_cv21_ta_30": 50.2, "gr_cv21_th_30": 55.95716198, "gr_cv21_vi_30": 81.56862745, "gr_cv21_zh_30": 68.1, "gr_emota_ta_30": 53.41880342, "gr_fleurs_en_30": 35.85780526, "gr_fleurs_km_30": 71.24183007, "gr_indowave_id_30": 49.33333333, "gr_m3ed_30": 48.7, "gr_openslr_ta_30": 49.8, "gr_sg_streets_utterance_30": 61.78861789, "gr_smaldusc_30": 49.6, "gr_thai_elderly_th_30": 31.55241935, "gr_thai_ser_th_30": 45.44502618, "gr_vietnam_celeb_30": 50.5, "gr_cv21_en_30": 55.3, "gr_sfdusc_30": 50.3 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 53.5746, "voxceleb_gender_test": 99.692, "iemocap_gender_test": 87.928, "gr_cv21_id_30": 46.6, "gr_cv21_ta_30": 50.4, "gr_cv21_th_30": 39.759, "gr_cv21_vi_30": 18.3007, "gr_cv21_zh_30": 34.4, "gr_emota_ta_30": 48.5043, "gr_fleurs_en_30": 67.8516, "gr_fleurs_km_30": 29.1503, "gr_indowave_id_30": 64.0, "gr_m3ed_30": 61.1, "gr_openslr_ta_30": 51.3, "gr_sg_streets_utterance_30": 40.4472, "gr_smaldusc_30": 50.1, "gr_thai_elderly_th_30": 60.7863, "gr_thai_ser_th_30": 66.0733, "gr_vietnam_celeb_30": 54.8, "gr_cv21_en_30": 50.1, "gr_sfdusc_30": 50.2 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 39.4927, "voxceleb_gender_test": 94.584, "iemocap_gender_test": 46.853, "gr_cv21_id_30": 33.2, "gr_cv21_ta_30": 42.2, "gr_cv21_th_30": 24.09638554, "gr_cv21_vi_30": 30.98039216, "gr_cv21_zh_30": 50.6, "gr_emota_ta_30": 31.78418803, "gr_fleurs_en_30": 50.54095827, "gr_fleurs_km_30": 22.35294118, "gr_indowave_id_30": 32.33333333, "gr_m3ed_30": 69.5, "gr_openslr_ta_30": 46.2, "gr_sg_streets_utterance_30": 39.83739837, "gr_smaldusc_30": 18.9, "gr_thai_elderly_th_30": 19.95967742, "gr_thai_ser_th_30": 41.83246073, "gr_vietnam_celeb_30": 36.6, "gr_cv21_en_30": 33.3, "gr_sfdusc_30": 24.2 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 39.3723, "voxceleb_gender_test": 32.786, "iemocap_gender_test": 62.948, "gr_cv21_id_30": 48.35, "gr_cv21_ta_30": 24.55, "gr_cv21_th_30": 47.12182062, "gr_cv21_vi_30": 47.90849673, "gr_cv21_zh_30": 71.0, "gr_emota_ta_30": 30.60897436, "gr_fleurs_en_30": 50.15455951, "gr_fleurs_km_30": 36.14379085, "gr_indowave_id_30": 17.33333333, "gr_m3ed_30": 16.95, "gr_openslr_ta_30": 39.2, "gr_sg_streets_utterance_30": 7.11382114, "gr_smaldusc_30": 68.2, "gr_thai_elderly_th_30": 46.40120968, "gr_thai_ser_th_30": 37.22513089, "gr_vietnam_celeb_30": 38.0, "gr_cv21_en_30": 24.35, "gr_sfdusc_30": 41.1 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "gr_cv21_th_30": 16.73360107, "gr_cv21_vi_30": 27.97385621, "gr_fleurs_en_30": 22.56568779, "gr_fleurs_km_30": 41.76470588, "gr_indowave_id_30": 36.66666667, "gr_openslr_ta_30": 22.4, "gr_sg_streets_utterance_30": 32.5203252, "gr_thai_elderly_th_30": 15.4233871, "gr_thai_ser_th_30": 36.28272251, "iemocap_gender_test": 36.95219124, "average": 30.8793, "gr_cv21_id_30": 23.2, "gr_cv21_ta_30": 23.55, "gr_cv21_zh_30": 39.6, "gr_emota_ta_30": 31.99786325, "gr_m3ed_30": 21.1, "gr_smaldusc_30": 34.2, "gr_vietnam_celeb_30": 33.7, "voxceleb_gender_test": 57.70414444, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT", "gr_cv21_en_30": 32.2, "gr_sfdusc_30": 31.05 } ], "metric": "accuracy / judge", "metricInfo": "Scale from 0-100. The higher, the better. AudioBench columns are Model-as-a-Judge (LLaMA-3-70B); AudioBench-SEA columns are accuracy.", "ascending": false } }, { "key": "Emotion Recognition", "title": "Task: Emotion Recognition", "taskName": "emotion_recognition", "metric": "accuracy / judge", "metricInfo": "Scale from 0-100. The higher, the better. AudioBench columns are Model-as-a-Judge (LLaMA-3-70B); AudioBench-SEA columns are accuracy.", "ascending": false, "datasets": [ { "display": "IEMOCAP-Emotion", "internal": "iemocap_emotion_test", "description": "Emotion recognition test data from the IEMOCAP dataset, focusing on identifying emotions in speech.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/iemocap_emotion_recognition", "stats": { "num_rows": 1004, "audio_length": { "min": 0.84, "max": 34.14, "mean": 4.42, "median": 3.42, "std": 3.26, "total_hours": 1.23, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 1003, 1, 0, 0, 0 ] } } } }, { "display": "MELD-Sentiment", "internal": "meld_sentiment_test", "description": "Sentiment recognition from speech using the MELD dataset, classifying positive, negative, or neutral sentiments.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/meld_sentiment_test", "stats": { "num_rows": 2610, "audio_length": { "min": 0.13, "max": 304.96, "mean": 3.35, "median": 2.54, "std": 7.8, "total_hours": 2.43, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 2608, 0, 1, 1, 0 ] } } } }, { "display": "MELD-Emotion", "internal": "meld_emotion_test", "description": "Emotion classification in speech using MELD, detecting specific emotions like happiness, anger, etc.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/meld_emotion_test", "stats": { "num_rows": 2610, "audio_length": { "min": 0.13, "max": 304.96, "mean": 3.35, "median": 2.54, "std": 7.8, "total_hours": 2.43, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 2608, 0, 1, 1, 0 ] } } } }, { "display": "ER-Emota-TA-30 [SEA]", "internal": "er_emota_ta_30", "description": "Emotion EMOTA Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "ER-ESD-EN-30 [SEA]", "internal": "er_esd_en_30", "description": "Emotion ESD English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "ER-ESD-ZH-30 [SEA]", "internal": "er_esd_zh_30", "description": "Emotion ESD Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "ER-Indowave-ID-30 [SEA]", "internal": "er_indowave_id_30", "description": "Emotion InDowave Indonesian 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "ER-M3ed-30 [SEA]", "internal": "er_m3ed_30", "description": "Emotion M3ED English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "ER-TEC-TA-30 [SEA]", "internal": "er_tec_ta_30", "description": "Emotion TEC Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "ER-THAI-SER-TH-30 [SEA]", "internal": "er_thai_ser_th_30", "description": "Emotion Thai SER Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} } ], "data": { "columns": [ { "key": "iemocap_emotion_test", "display": "IEMOCAP-Emotion", "description": "Emotion recognition test data from the IEMOCAP dataset, focusing on identifying emotions in speech.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/iemocap_emotion_recognition", "isWer": false }, { "key": "meld_sentiment_test", "display": "MELD-Sentiment", "description": "Sentiment recognition from speech using the MELD dataset, classifying positive, negative, or neutral sentiments.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/meld_sentiment_test", "isWer": false }, { "key": "meld_emotion_test", "display": "MELD-Emotion", "description": "Emotion classification in speech using MELD, detecting specific emotions like happiness, anger, etc.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/meld_emotion_test", "isWer": false }, { "key": "er_emota_ta_30", "display": "ER-Emota-TA-30 [SEA]", "description": "Emotion EMOTA Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "er_esd_en_30", "display": "ER-ESD-EN-30 [SEA]", "description": "Emotion ESD English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "er_esd_zh_30", "display": "ER-ESD-ZH-30 [SEA]", "description": "Emotion ESD Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "er_indowave_id_30", "display": "ER-Indowave-ID-30 [SEA]", "description": "Emotion InDowave Indonesian 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "er_m3ed_30", "display": "ER-M3ed-30 [SEA]", "description": "Emotion M3ED English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "er_tec_ta_30", "display": "ER-TEC-TA-30 [SEA]", "description": "Emotion TEC Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "er_thai_ser_th_30", "display": "ER-THAI-SER-TH-30 [SEA]", "description": "Emotion Thai SER Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false } ], "rows": [ { "model": "MERaLiON-SER-v1", "er_esd_zh_30": 97.4, "average": 54.3879, "er_esd_en_30": 95.0, "er_indowave_id_30": 43.33333333, "er_m3ed_30": 41.75, "er_thai_ser_th_30": 75.18324607, "meld_sentiment_test": 29.55938697, "er_emota_ta_30": 30.98290598, "er_tec_ta_30": 61.21212121, "iemocap_emotion_test": 39.34262948, "meld_emotion_test": 30.11494253, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SER-v1" }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 36.1017, "iemocap_emotion_test": 53.685, "meld_sentiment_test": 59.847, "meld_emotion_test": 47.548, "er_emota_ta_30": 22.54273504, "er_esd_en_30": 30.6, "er_esd_zh_30": 28.8, "er_indowave_id_30": 20.0, "er_m3ed_30": 29.3, "er_tec_ta_30": 48.48484848, "er_thai_ser_th_30": 20.20942408 }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 33.8733, "iemocap_emotion_test": 59.761, "meld_sentiment_test": 51.073, "meld_emotion_test": 41.571, "er_emota_ta_30": 24.25213675, "er_esd_en_30": 20.65, "er_esd_zh_30": 17.9, "er_indowave_id_30": 19.0, "er_m3ed_30": 24.85, "er_tec_ta_30": 54.54545455, "er_thai_ser_th_30": 25.13089005 }, { "model": "MERaLiON-3-3B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR", "er_esd_en_30": 39.0, "er_m3ed_30": 43.6, "er_tec_ta_30": 33.33333333, "er_thai_ser_th_30": 38.32460733, "iemocap_emotion_test": 42.43027888, "meld_emotion_test": 31.41762452, "average": 33.8309, "er_esd_zh_30": 38.8, "er_indowave_id_30": 26.66666667, "er_emota_ta_30": 18.37606838, "meld_sentiment_test": 26.36015326 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 32.8289, "iemocap_emotion_test": 62.55, "meld_sentiment_test": 68.851, "meld_emotion_test": 59.808, "er_emota_ta_30": 9.2415, "er_esd_en_30": 20.05, "er_esd_zh_30": 15.025, "er_indowave_id_30": 17.3333, "er_m3ed_30": 21.2, "er_tec_ta_30": 40.303, "er_thai_ser_th_30": 13.9267 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 32.7288, "iemocap_emotion_test": 51.394, "meld_sentiment_test": 58.582, "meld_emotion_test": 52.146, "er_emota_ta_30": 16.0256, "er_esd_en_30": 23.75, "er_esd_zh_30": 20.85, "er_indowave_id_30": 20.0, "er_m3ed_30": 26.65, "er_tec_ta_30": 37.5758, "er_thai_ser_th_30": 20.3141 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 31.5524, "iemocap_emotion_test": 49.104, "meld_sentiment_test": 52.452, "meld_emotion_test": 44.176, "er_emota_ta_30": 23.71794872, "er_esd_en_30": 21.45, "er_esd_zh_30": 18.65, "er_indowave_id_30": 20.66666667, "er_m3ed_30": 21.2, "er_tec_ta_30": 43.63636364, "er_thai_ser_th_30": 20.47120419 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 31.4805, "iemocap_emotion_test": 53.984, "meld_sentiment_test": 53.946, "meld_emotion_test": 41.609, "er_esd_zh_30": 32.0, "er_indowave_id_30": 12.0, "er_tec_ta_30": 30.3030303, "er_emota_ta_30": 22.48931624, "er_esd_en_30": 41.8, "er_m3ed_30": 11.7, "er_thai_ser_th_30": 14.97382199 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 30.8805, "iemocap_emotion_test": 29.382, "meld_sentiment_test": 44.904, "meld_emotion_test": 50.728, "er_emota_ta_30": 21.68803419, "er_esd_en_30": 49.2, "er_esd_zh_30": 18.5, "er_indowave_id_30": 21.16666667, "er_m3ed_30": 16.1, "er_tec_ta_30": 42.42424242, "er_thai_ser_th_30": 14.71204188 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 26.1694, "iemocap_emotion_test": 32.072, "meld_sentiment_test": 49.119, "meld_emotion_test": 40.843, "er_emota_ta_30": 19.65811966, "er_esd_en_30": 21.95, "er_esd_zh_30": 17.45, "er_indowave_id_30": 18.0, "er_m3ed_30": 22.2, "er_tec_ta_30": 23.33333333, "er_thai_ser_th_30": 17.06806283 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 24.2284, "iemocap_emotion_test": 51.693, "meld_sentiment_test": 50.805, "meld_emotion_test": 53.525, "er_emota_ta_30": 8.01282051, "er_esd_en_30": 13.1, "er_esd_zh_30": 7.55, "er_indowave_id_30": 11.33333333, "er_m3ed_30": 9.7, "er_tec_ta_30": 25.15151515, "er_thai_ser_th_30": 11.41361257 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 22.1378, "iemocap_emotion_test": 26.195, "meld_sentiment_test": 42.261, "meld_emotion_test": 32.299, "er_emota_ta_30": 15.11752137, "er_esd_en_30": 15.95, "er_esd_zh_30": 13.1, "er_indowave_id_30": 10.33333333, "er_m3ed_30": 16.4, "er_tec_ta_30": 27.57575758, "er_thai_ser_th_30": 22.14659686 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 20.979, "iemocap_emotion_test": 36.554, "meld_sentiment_test": 27.778, "meld_emotion_test": 30.077, "er_emota_ta_30": 13.30128205, "er_esd_en_30": 11.05, "er_esd_zh_30": 17.9, "er_indowave_id_30": 17.0, "er_m3ed_30": 13.95, "er_tec_ta_30": 29.09090909, "er_thai_ser_th_30": 13.08900524 }, { "model": "MERaLiON-3-10B", "er_emota_ta_30": 10.47008547, "er_esd_en_30": 18.4, "er_esd_zh_30": 13.95, "er_indowave_id_30": 16.0, "er_m3ed_30": 16.5, "er_tec_ta_30": 46.96969697, "er_thai_ser_th_30": 10.2617801, "iemocap_emotion_test": 25.64741036, "average": 20.4233, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B", "meld_emotion_test": 19.92337165, "meld_sentiment_test": 26.11111111 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 19.3616, "iemocap_emotion_test": 34.363, "meld_sentiment_test": 30.421, "meld_emotion_test": 34.33, "er_emota_ta_30": 5.44871795, "er_esd_en_30": 10.8, "er_esd_zh_30": 15.4, "er_indowave_id_30": 13.66666667, "er_m3ed_30": 10.9, "er_tec_ta_30": 28.18181818, "er_thai_ser_th_30": 10.10471204 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "er_emota_ta_30": 16.6667, "er_esd_en_30": 17.35, "er_esd_zh_30": 14.95, "er_indowave_id_30": 14.1667, "er_m3ed_30": 17.05, "er_tec_ta_30": 34.8485, "er_thai_ser_th_30": 8.8482, "average": 17.6972 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "er_emota_ta_30": 13.46153846, "er_esd_zh_30": 8.65, "er_indowave_id_30": 12.33333333, "er_m3ed_30": 8.95, "er_tec_ta_30": 17.87878788, "meld_emotion_test": 13.63984674, "meld_sentiment_test": 27.62452107, "average": 14.9305, "er_esd_en_30": 9.15, "iemocap_emotion_test": 26.04581673, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT", "er_thai_ser_th_30": 11.57068063 } ], "metric": "accuracy / judge", "metricInfo": "Scale from 0-100. The higher, the better. AudioBench columns are Model-as-a-Judge (LLaMA-3-70B); AudioBench-SEA columns are accuracy.", "ascending": false } }, { "key": "Music Understanding", "title": "Task: Music Understanding - MCQ Questions", "taskName": "music_understanding", "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false, "datasets": [ { "display": "MuChoMusic", "internal": "muchomusic_test", "description": "Test dataset for music understanding, from paper: MuChoMusic: Evaluating Music Understanding in Multimodal Audio-Language Models.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/mu_chomusic_test", "stats": { "num_rows": 1187, "audio_length": { "min": 9.89, "max": 119.99, "mean": 40.35, "median": 10.01, "std": 49.05, "total_hours": 13.31, "histogram": { "bin_edges": [ 0, 30, 120, 300, 1800 ], "bin_labels": [ "0-30s", "30-120s", "120-300s", "300-1800s", "1800s+" ], "counts": [ 858, 329, 0, 0, 0 ] } } } } ], "data": { "columns": [ { "key": "muchomusic_test", "display": "MuChoMusic", "description": "Test dataset for music understanding, from paper: MuChoMusic: Evaluating Music Understanding in Multimodal Audio-Language Models.", "hfLink": "https://huggingface.co/datasets/AudioLLMs/mu_chomusic_test", "isWer": false } ], "rows": [ { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://arxiv.org/abs/2407.10759", "average": 71.609, "muchomusic_test": 71.609 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "average": 63.943, "muchomusic_test": 63.943 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "average": 63.69, "muchomusic_test": 63.69 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "average": 60.657, "muchomusic_test": 60.657 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "average": 59.309, "muchomusic_test": 59.309 }, { "model": "Qwen-Audio-Chat", "modelLink": "https://arxiv.org/abs/2311.07919", "average": 59.056, "muchomusic_test": 59.056 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "average": 55.602, "muchomusic_test": 55.602 }, { "model": "MERaLiON-3-10B", "muchomusic_test": 55.3665, "average": 55.3665, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "average": 55.265, "muchomusic_test": 55.265 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION", "average": 51.348, "muchomusic_test": 51.348 }, { "model": "SALMONN-7B", "modelLink": "https://arxiv.org/html/2310.13289v2", "average": 49.705, "muchomusic_test": 49.705 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "average": 47.599, "muchomusic_test": 47.599 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "muchomusic_test": 44.8526, "average": 44.8526, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "MERaLiON-3-3B-ASR", "muchomusic_test": 44.7852, "average": 44.7852, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "WavLLM", "modelLink": "https://arxiv.org/abs/2404.00656", "average": 44.313, "muchomusic_test": 44.313 } ], "metric": "llama3_70b_judge", "metricInfo": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "ascending": false } }, { "key": "ASR-Burmese", "title": "Task: Automatic Speech Recognition - Burmese", "taskName": "asr_burmese", "metric": "cer", "metricInfo": "Character Error Rate (CER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "Bloomspeech-MY-30 [SEA]", "internal": "asr_bloomspeech_my_30", "description": "BloomSpeech Burmese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Fleurs-MY-30 [SEA]", "internal": "asr_fleurs_my_30", "description": "Fleurs Burmese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "MIG-MY-30 [SEA]", "internal": "asr_mig_my_30", "description": "MIG Burmese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Openslr-MY-30 [SEA]", "internal": "asr_openslr_my_30", "description": "OpenSLR Burmese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} } ], "data": { "columns": [ { "key": "asr_bloomspeech_my_30", "display": "Bloomspeech-MY-30 [SEA]", "description": "BloomSpeech Burmese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_fleurs_my_30", "display": "Fleurs-MY-30 [SEA]", "description": "Fleurs Burmese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_mig_my_30", "display": "MIG-MY-30 [SEA]", "description": "MIG Burmese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_openslr_my_30", "display": "Openslr-MY-30 [SEA]", "description": "OpenSLR Burmese ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true } ], "rows": [ { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_bloomspeech_my_30": 0.70794437, "asr_fleurs_my_30": 0.36029653, "asr_mig_my_30": 0.45294897, "asr_openslr_my_30": 0.29198099, "average": 0.4533 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_bloomspeech_my_30": 0.77481425, "asr_fleurs_my_30": 0.35027339, "asr_mig_my_30": 1.17876077, "asr_openslr_my_30": 0.27381126, "average": 0.6444, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "asr_bloomspeech_my_30": 0.7953896, "asr_fleurs_my_30": 0.61972925, "asr_mig_my_30": 0.95311465, "asr_openslr_my_30": 0.52812252, "average": 0.7241 }, { "model": "MERaLiON-3-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B", "asr_bloomspeech_my_30": 0.86473614, "asr_mig_my_30": 1.0054672, "asr_openslr_my_30": 0.575116, "average": 0.8151 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "asr_bloomspeech_my_30": 0.81634597, "asr_fleurs_my_30": 0.79148354, "asr_mig_my_30": 1.03462558, "asr_openslr_my_30": 0.6989666, "average": 0.8354 }, { "model": "MERaLiON-3-3B-ASR-CTM", "asr_bloomspeech_my_30": 0.96818442, "asr_fleurs_my_30": 0.86464763, "asr_mig_my_30": 1.0102717, "asr_openslr_my_30": 0.6228463, "average": 0.8665 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "asr_bloomspeech_my_30": 0.85921128, "asr_mig_my_30": 1.12292909, "average": 0.9283, "asr_fleurs_my_30": 0.84420908, "asr_openslr_my_30": 0.88666083, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION" }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 0.983, "asr_mig_my_30": 1.38518887, "asr_bloomspeech_my_30": 0.81729853, "asr_fleurs_my_30": 0.82671662, "asr_openslr_my_30": 0.902625 }, { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_bloomspeech_my_30": 1.0, "asr_fleurs_my_30": 1.00064846, "asr_mig_my_30": 1.00165673, "asr_openslr_my_30": 1.00374769, "average": 1.0015 }, { "model": "Fun-ASR-MLT-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512", "asr_bloomspeech_my_30": 1.0, "asr_fleurs_my_30": 1.05890594, "asr_mig_my_30": 1.02054341, "asr_openslr_my_30": 1.03353288, "average": 1.0282 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "asr_bloomspeech_my_30": 1.0, "asr_mig_my_30": 1.00944334, "average": 1.0349, "asr_fleurs_my_30": 1.11261618, "asr_openslr_my_30": 1.01747214 }, { "model": "MERaLiON-3-3B-ASR", "asr_bloomspeech_my_30": 1.0672509, "asr_fleurs_my_30": 1.03252716, "asr_mig_my_30": 1.12110669, "asr_openslr_my_30": 0.9327363, "average": 1.0384, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "asr_bloomspeech_my_30": 1.04743761, "asr_fleurs_my_30": 0.97222823, "asr_mig_my_30": 1.51905235, "asr_openslr_my_30": 0.65135218, "average": 1.0475 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "asr_bloomspeech_my_30": 1.01562202, "asr_fleurs_my_30": 1.1138, "asr_mig_my_30": 1.06312127, "asr_openslr_my_30": 1.0877, "average": 1.0701 }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_bloomspeech_my_30": 1.00076205, "asr_fleurs_my_30": 1.11444148, "asr_mig_my_30": 1.21868787, "asr_openslr_my_30": 1.05103994, "average": 1.0962 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "asr_bloomspeech_my_30": 1.65174319, "asr_fleurs_my_30": 0.62212295, "asr_openslr_my_30": 0.76348534, "average": 1.1012, "asr_mig_my_30": 1.36729622 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "asr_bloomspeech_my_30": 1.0, "asr_fleurs_my_30": 1.0, "asr_mig_my_30": 2.04854208, "asr_openslr_my_30": 1.0, "average": 1.2621 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "asr_bloomspeech_my_30": 1.00019051, "asr_fleurs_my_30": 1.5961, "asr_mig_my_30": 1.57670643, "asr_openslr_my_30": 1.1845, "average": 1.3394 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 1.4864, "asr_bloomspeech_my_30": 0.85101924, "asr_mig_my_30": 3.29456594, "asr_openslr_my_30": 1.01198877, "asr_fleurs_my_30": 0.78795302 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_fleurs_my_30": 1.09627655, "asr_mig_my_30": 3.25215374, "asr_openslr_my_30": 1.22335783, "average": 1.7427, "asr_bloomspeech_my_30": 1.39893313 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_fleurs_my_30": 1.53524509, "asr_mig_my_30": 2.14413519, "asr_openslr_my_30": 2.03311269, "average": 1.7537, "asr_bloomspeech_my_30": 1.3023433 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "asr_bloomspeech_my_30": 1.4022, "asr_fleurs_my_30": 1.2516, "asr_mig_my_30": 2.5306, "asr_openslr_my_30": 2.3581, "average": 1.8856 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "asr_bloomspeech_my_30": 1.1343113, "asr_fleurs_my_30": 2.6527, "asr_mig_my_30": 2.70526839, "asr_openslr_my_30": 3.1073, "average": 2.3999 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "asr_bloomspeech_my_30": 2.59878072, "asr_fleurs_my_30": 3.2622, "asr_mig_my_30": 7.13833665, "asr_openslr_my_30": 6.3163, "average": 4.8289 } ], "metric": "cer", "metricInfo": "Character Error Rate (CER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "ASR-Tagalog", "title": "Task: Automatic Speech Recognition - Tagalog", "taskName": "asr_tagalog", "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "Bloomspeech-TL-30 [SEA]", "internal": "asr_bloomspeech_tl_30", "description": "BloomSpeech Tagalog ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Fleurs-TL-30 [SEA]", "internal": "asr_fleurs_tl_30", "description": "Fleurs Tagalog ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Sfdusc-30 [SEA]", "internal": "asr_sfdusc_30", "description": "SFDUSC Tagalog ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} } ], "data": { "columns": [ { "key": "asr_bloomspeech_tl_30", "display": "Bloomspeech-TL-30 [SEA]", "description": "BloomSpeech Tagalog ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_fleurs_tl_30", "display": "Fleurs-TL-30 [SEA]", "description": "Fleurs Tagalog ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_sfdusc_30", "display": "Sfdusc-30 [SEA]", "description": "SFDUSC Tagalog ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true } ], "rows": [ { "model": "MERaLiON-3-3B-ASR-CTM", "asr_bloomspeech_tl_30": 0.10775862, "asr_fleurs_tl_30": 0.12505216, "asr_sfdusc_30": 0.1985225, "average": 0.1438 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "asr_bloomspeech_tl_30": 0.1164, "asr_fleurs_tl_30": 0.1183, "asr_sfdusc_30": 0.2107, "average": 0.1485 }, { "model": "MERaLiON-3-3B-ASR", "asr_bloomspeech_tl_30": 0.12284483, "asr_fleurs_tl_30": 0.12200618, "asr_sfdusc_30": 0.20591001, "average": 0.1503, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "asr_bloomspeech_tl_30": 0.10775862, "asr_fleurs_tl_30": 0.12555287, "asr_sfdusc_30": 0.22243116, "average": 0.1519 }, { "model": "MERaLiON-3-10B", "asr_fleurs_tl_30": 0.14220145, "asr_sfdusc_30": 0.21302888, "average": 0.155, "asr_bloomspeech_tl_30": 0.10991379, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "asr_bloomspeech_tl_30": 0.12931034, "asr_fleurs_tl_30": 0.15471919, "asr_sfdusc_30": 0.23451981, "average": 0.1728 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "asr_bloomspeech_tl_30": 0.14439655, "asr_fleurs_tl_30": 0.1748, "asr_sfdusc_30": 0.2658, "average": 0.195 }, { "model": "Fun-ASR-MLT-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512", "asr_bloomspeech_tl_30": 0.11422414, "asr_fleurs_tl_30": 0.14587332, "asr_sfdusc_30": 0.33646743, "average": 0.1989 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_bloomspeech_tl_30": 0.20258621, "asr_fleurs_tl_30": 0.1608946, "asr_sfdusc_30": 0.33754197, "average": 0.2337 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_bloomspeech_tl_30": 0.17241379, "asr_fleurs_tl_30": 0.20078444, "asr_sfdusc_30": 0.47360645, "average": 0.2823 }, { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_bloomspeech_tl_30": 0.22844828, "asr_fleurs_tl_30": 0.24776767, "asr_sfdusc_30": 0.38509066, "average": 0.2871 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "asr_bloomspeech_tl_30": 0.21336207, "asr_fleurs_tl_30": 0.15484436, "asr_sfdusc_30": 0.59314976, "average": 0.3205 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_bloomspeech_tl_30": 0.62715517, "asr_fleurs_tl_30": 0.22556956, "asr_sfdusc_30": 0.52411014, "average": 0.4589, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "asr_bloomspeech_tl_30": 0.45258621, "average": 0.5285, "asr_fleurs_tl_30": 0.42422599, "asr_sfdusc_30": 0.70866353, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION" }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_bloomspeech_tl_30": 0.49784483, "asr_fleurs_tl_30": 0.20420596, "asr_sfdusc_30": 0.92155809, "average": 0.5412 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "asr_bloomspeech_tl_30": 0.4137931, "asr_fleurs_tl_30": 0.7151, "asr_sfdusc_30": 0.5698, "average": 0.5662 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 0.6129, "asr_bloomspeech_tl_30": 0.53232759, "asr_sfdusc_30": 0.76964406, "asr_fleurs_tl_30": 0.53671868 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "asr_bloomspeech_tl_30": 0.51724138, "asr_fleurs_tl_30": 1.2921, "asr_sfdusc_30": 0.7302, "average": 0.8465 }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "asr_fleurs_tl_30": 0.85425186, "asr_sfdusc_30": 0.85399597, "average": 0.851, "asr_bloomspeech_tl_30": 0.84482759, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2" }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_bloomspeech_tl_30": 1.02155172, "asr_fleurs_tl_30": 0.90361345, "asr_sfdusc_30": 0.97609134, "average": 0.9671 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "asr_bloomspeech_tl_30": 1.06681034, "asr_fleurs_tl_30": 1.01493783, "asr_sfdusc_30": 1.1128274, "average": 1.0649 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "asr_bloomspeech_tl_30": 0.87931034, "asr_fleurs_tl_30": 1.5243, "asr_sfdusc_30": 1.2414, "average": 1.215 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "asr_bloomspeech_tl_30": 1.43318966, "asr_fleurs_tl_30": 1.3328, "asr_sfdusc_30": 2.1392, "average": 1.6351 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 2.2607, "asr_bloomspeech_tl_30": 2.62284483, "asr_fleurs_tl_30": 1.27217725, "asr_sfdusc_30": 2.8871726 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "asr_bloomspeech_tl_30": 5.38577586, "asr_fleurs_tl_30": 1.0, "asr_sfdusc_30": 1.0, "average": 2.4619 } ], "metric": "wer", "metricInfo": "Word Error Rate (WER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "ASR-Khmer", "title": "Task: Automatic Speech Recognition - Khmer", "taskName": "asr_khmer", "metric": "cer", "metricInfo": "Character Error Rate (CER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "Fleurs-KM-30 [SEA]", "internal": "asr_fleurs_km_30", "description": "Fleurs Khmer ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} }, { "display": "Openslr-KM-30 [SEA]", "internal": "asr_openslr_km_30", "description": "OpenSLR Khmer ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} } ], "data": { "columns": [ { "key": "asr_fleurs_km_30", "display": "Fleurs-KM-30 [SEA]", "description": "Fleurs Khmer ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true }, { "key": "asr_openslr_km_30", "display": "Openslr-KM-30 [SEA]", "description": "OpenSLR Khmer ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true } ], "rows": [ { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_fleurs_km_30": 0.32348064, "asr_openslr_km_30": 0.26768711, "average": 0.2956 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_fleurs_km_30": 0.33927738, "asr_openslr_km_30": 0.52739954, "average": 0.4333, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "asr_fleurs_km_30": 0.50452257, "asr_openslr_km_30": 1.02682822, "average": 0.7657 }, { "model": "MERaLiON-AudioLLM-Whisper-SEA-LION", "asr_openslr_km_30": 0.88818796, "average": 0.8919, "asr_fleurs_km_30": 0.89568792, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-AudioLLM-Whisper-SEA-LION" }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "asr_fleurs_km_30": 0.98882016, "asr_openslr_km_30": 1.01394972, "average": 1.0014, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "asr_fleurs_km_30": 0.8379, "asr_openslr_km_30": 1.23062274, "average": 1.0343 }, { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_fleurs_km_30": 1.0361508, "asr_openslr_km_30": 1.04720529, "average": 1.0417 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "asr_fleurs_km_30": 0.9885, "asr_openslr_km_30": 1.10369453, "average": 1.0461 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "asr_fleurs_km_30": 1.07066334, "asr_openslr_km_30": 1.08007998, "average": 1.0754 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_fleurs_km_30": 0.75539383, "asr_openslr_km_30": 1.46367359, "average": 1.1095 }, { "model": "MERaLiON-3-3B-ASR", "asr_fleurs_km_30": 1.15897018, "asr_openslr_km_30": 1.12266711, "average": 1.1408, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "MERaLiON-3-3B-ASR-CTM", "asr_fleurs_km_30": 1.1827894, "asr_openslr_km_30": 1.12223862, "average": 1.1525 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "asr_fleurs_km_30": 1.16334879, "asr_openslr_km_30": 1.1421872, "average": 1.1528 }, { "model": "Fun-ASR-MLT-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-MLT-Nano-2512", "asr_fleurs_km_30": 1.17109326, "average": 1.1577, "asr_openslr_km_30": 1.14437726 }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_fleurs_km_30": 1.24152824, "asr_openslr_km_30": 1.11600171, "average": 1.1788 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_fleurs_km_30": 1.67597327, "asr_openslr_km_30": 0.79875262, "average": 1.2374 }, { "model": "MERaLiON-3-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B", "asr_fleurs_km_30": 1.16583099, "asr_openslr_km_30": 1.3358646, "average": 1.2508 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "asr_fleurs_km_30": 1.3673, "asr_openslr_km_30": 1.15901733, "average": 1.2632 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 1.3407, "asr_openslr_km_30": 1.66423062, "asr_fleurs_km_30": 1.01708748 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "asr_fleurs_km_30": 1.7615, "asr_openslr_km_30": 1.7648, "average": 1.7631 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 2.0216, "asr_fleurs_km_30": 1.15895033, "asr_openslr_km_30": 2.88421253 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "asr_fleurs_km_30": 3.1356, "asr_openslr_km_30": 1.94072558, "average": 2.5382 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "asr_fleurs_km_30": 1.0, "asr_openslr_km_30": 4.09019711, "average": 2.5451 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "asr_fleurs_km_30": 1.2804, "asr_openslr_km_30": 4.3235, "average": 2.802 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "asr_fleurs_km_30": 4.3865, "asr_openslr_km_30": 3.01354504, "average": 3.7 } ], "metric": "cer", "metricInfo": "Character Error Rate (CER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "ASR-Lao", "title": "Task: Automatic Speech Recognition - Lao", "taskName": "asr_lao", "metric": "cer", "metricInfo": "Character Error Rate (CER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "Fleurs-LO-30 [SEA]", "internal": "asr_fleurs_lo_30", "description": "Fleurs Lao ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "stats": {} } ], "data": { "columns": [ { "key": "asr_fleurs_lo_30", "display": "Fleurs-LO-30 [SEA]", "description": "Fleurs Lao ASR 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_ASR", "isWer": true } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "asr_fleurs_lo_30": 0.28207057, "average": 0.2821 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "asr_fleurs_lo_30": 0.35719197, "average": 0.3572, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "asr_fleurs_lo_30": 0.3834, "average": 0.3834 }, { "model": "MERaLiON-2-10B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B-ASR", "asr_fleurs_lo_30": 0.3851, "average": 0.3851 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "asr_fleurs_lo_30": 0.49359663, "average": 0.4936 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "asr_fleurs_lo_30": 0.73168164, "average": 0.7317 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "asr_fleurs_lo_30": 0.82562637, "average": 0.8256 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "asr_fleurs_lo_30": 0.95279127, "average": 0.9528 }, { "model": "MERaLiON-SpeechEncoder2-ASR-CTC", "asr_fleurs_lo_30": 0.99939834, "average": 0.9994, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-SpeechEncoder-2" }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "asr_fleurs_lo_30": 1.0, "average": 1.0 }, { "model": "Whisper-large-v3", "modelLink": "https://huggingface.co/openai/whisper-large-v3", "asr_fleurs_lo_30": 1.0133, "average": 1.0133 }, { "model": "Qwen3-ASR-1.7B", "modelLink": "https://huggingface.co/Qwen/Qwen3-ASR-1.7B", "asr_fleurs_lo_30": 1.01813572, "average": 1.0181 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "asr_fleurs_lo_30": 1.0234, "average": 1.0234 }, { "model": "MERaLiON-3-3B-ASR", "asr_fleurs_lo_30": 1.03577721, "average": 1.0358, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR" }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "asr_fleurs_lo_30": 1.0362, "average": 1.0362 }, { "model": "MERaLiON-3-3B-ASR-CTM", "asr_fleurs_lo_30": 1.03760368, "average": 1.0376 }, { "model": "MERaLiON-3-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B", "asr_fleurs_lo_30": 1.045855, "average": 1.0459 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "average": 1.0511, "asr_fleurs_lo_30": 1.051141 }, { "model": "Fun-ASR-Nano-2512", "modelLink": "https://huggingface.co/FunAudioLLM/Fun-ASR-Nano-2512", "asr_fleurs_lo_30": 1.06010142, "average": 1.0601 }, { "model": "canary-qwen-2.5b", "modelLink": "https://huggingface.co/nvidia/canary-qwen-2.5b", "asr_fleurs_lo_30": 1.1246938, "average": 1.1247 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "asr_fleurs_lo_30": 1.2216, "average": 1.2216 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "average": 1.3439, "asr_fleurs_lo_30": 1.34393399 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "asr_fleurs_lo_30": 1.4651, "average": 1.4651 } ], "metric": "cer", "metricInfo": "Character Error Rate (CER) - The Lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "SQA-Mandarin", "title": "Task: SQA - Mandarin", "taskName": "sqa_mandarin", "metric": "judge", "metricInfo": "Model-as-judge correctness score - The higher, the better.", "ascending": false, "datasets": [ { "display": "Sgpccsc-30 [SEA]", "internal": "sqa_sgpccsc_30", "description": "SQ PCCSC Chinese SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "stats": {} }, { "display": "Yodas2-ZH-30 [SEA]", "internal": "sqa_yodas2_zh_30", "description": "YODAS2 Chinese SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "stats": {} } ], "data": { "columns": [ { "key": "sqa_sgpccsc_30", "display": "Sgpccsc-30 [SEA]", "description": "SQ PCCSC Chinese SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "isWer": false }, { "key": "sqa_yodas2_zh_30", "display": "Yodas2-ZH-30 [SEA]", "description": "YODAS2 Chinese SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "sqa_sgpccsc_30": 88.6146, "sqa_yodas2_zh_30": 87.163, "average": 87.8888 }, { "model": "MERaLiON-3-10B", "sqa_sgpccsc_30": 86.54911839, "sqa_yodas2_zh_30": 80.8249497, "average": 83.687, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "sqa_sgpccsc_30": 84.1814, "sqa_yodas2_zh_30": 77.6459, "average": 80.9136 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "sqa_sgpccsc_30": 70.932, "sqa_yodas2_zh_30": 55.6539, "average": 63.293 }, { "model": "MERaLiON-3-3B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR", "sqa_yodas2_zh_30": 57.72635815, "average": 57.7264 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "sqa_sgpccsc_30": 48.91687657, "sqa_yodas2_zh_30": 37.98792757, "average": 43.4524, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" } ], "metric": "judge", "metricInfo": "Model-as-judge correctness score - The higher, the better.", "ascending": false } }, { "key": "SQA-Indonesian", "title": "Task: SQA - Indonesian", "taskName": "sqa_indonesian", "metric": "judge", "metricInfo": "Model-as-judge correctness score - The higher, the better.", "ascending": false, "datasets": [ { "display": "Yodas2-ID-30 [SEA]", "internal": "sqa_yodas2_id_30", "description": "YODAS2 Indonesian SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "stats": {} } ], "data": { "columns": [ { "key": "sqa_yodas2_id_30", "display": "Yodas2-ID-30 [SEA]", "description": "YODAS2 Indonesian SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "sqa_yodas2_id_30": 84.4064, "average": 84.4064 }, { "model": "MERaLiON-3-10B", "sqa_yodas2_id_30": 83.76257545, "average": 83.7626, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "sqa_yodas2_id_30": 81.4688, "average": 81.4688 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "sqa_yodas2_id_30": 65.2113, "average": 65.2113 }, { "model": "MERaLiON-3-3B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR", "sqa_yodas2_id_30": 63.48088531, "average": 63.4809 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "sqa_yodas2_id_30": 49.22535211, "average": 49.2254, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" } ], "metric": "judge", "metricInfo": "Model-as-judge correctness score - The higher, the better.", "ascending": false } }, { "key": "SQA-Thai", "title": "Task: SQA - Thai", "taskName": "sqa_thai", "metric": "judge", "metricInfo": "Model-as-judge correctness score - The higher, the better.", "ascending": false, "datasets": [ { "display": "Yodas2-TH-30 [SEA]", "internal": "sqa_yodas2_th_30", "description": "YODAS2 Thai SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "stats": {} } ], "data": { "columns": [ { "key": "sqa_yodas2_th_30", "display": "Yodas2-TH-30 [SEA]", "description": "YODAS2 Thai SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "sqa_yodas2_th_30": 88.0762, "average": 88.0762 }, { "model": "MERaLiON-3-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B", "sqa_yodas2_th_30": 80.6012024, "average": 80.6012 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "sqa_yodas2_th_30": 78.6974, "average": 78.6974 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "sqa_yodas2_th_30": 68.33667335, "average": 68.3367, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "sqa_yodas2_th_30": 46.4529, "average": 46.4529 } ], "metric": "judge", "metricInfo": "Model-as-judge correctness score - The higher, the better.", "ascending": false } }, { "key": "SQA-Vietnamese", "title": "Task: SQA - Vietnamese", "taskName": "sqa_vietnamese", "metric": "judge", "metricInfo": "Model-as-judge correctness score - The higher, the better.", "ascending": false, "datasets": [ { "display": "Yodas2-VI-30 [SEA]", "internal": "sqa_yodas2_vi_30", "description": "YODAS2 Vietnamese SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "stats": {} } ], "data": { "columns": [ { "key": "sqa_yodas2_vi_30", "display": "Yodas2-VI-30 [SEA]", "description": "YODAS2 Vietnamese SQA 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SQA", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "sqa_yodas2_vi_30": 81.9153, "average": 81.9153 }, { "model": "MERaLiON-3-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B", "sqa_yodas2_vi_30": 76.37096774, "average": 76.371 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "sqa_yodas2_vi_30": 69.2742, "average": 69.2742 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "sqa_yodas2_vi_30": 42.9234, "average": 42.9234 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "sqa_yodas2_vi_30": 39.5766129, "average": 39.5766, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" } ], "metric": "judge", "metricInfo": "Model-as-judge correctness score - The higher, the better.", "ascending": false } }, { "key": "TCQ-English", "title": "Task: TCQ - English", "taskName": "tcq_english", "metric": "wer", "metricInfo": "Word/Character Error Rate over the queried time interval - The lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "TCQ-SG-Streets-30 [SEA]", "internal": "tcq_sg_streets_30", "description": "TCQ SG Streets Singapore English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-SG-Streets-60 [SEA]", "internal": "tcq_sg_streets_60", "description": "TCQ SG Streets Singapore English 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-EN-30 [SEA]", "internal": "tcq_yodas2_en_30", "description": "TCQ YODAS2 English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-EN-60 [SEA]", "internal": "tcq_yodas2_en_60", "description": "TCQ YODAS2 English 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-EN-120 [SEA]", "internal": "tcq_yodas2_en_120", "description": "TCQ YODAS2 English 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-EN-180 [SEA]", "internal": "tcq_yodas2_en_180", "description": "TCQ YODAS2 English 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-SG-Streets-120 [SEA]", "internal": "tcq_sg_streets_120", "description": "TCQ SG Streets Singapore English 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/MERaLiON/sea_audiobench_datasets_TCQ", "stats": {} } ], "data": { "columns": [ { "key": "tcq_sg_streets_30", "display": "TCQ-SG-Streets-30 [SEA]", "description": "TCQ SG Streets Singapore English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_sg_streets_60", "display": "TCQ-SG-Streets-60 [SEA]", "description": "TCQ SG Streets Singapore English 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_en_30", "display": "TCQ-Yodas2-EN-30 [SEA]", "description": "TCQ YODAS2 English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_en_60", "display": "TCQ-Yodas2-EN-60 [SEA]", "description": "TCQ YODAS2 English 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_en_120", "display": "TCQ-Yodas2-EN-120 [SEA]", "description": "TCQ YODAS2 English 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_en_180", "display": "TCQ-Yodas2-EN-180 [SEA]", "description": "TCQ YODAS2 English 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_sg_streets_120", "display": "TCQ-SG-Streets-120 [SEA]", "description": "TCQ SG Streets Singapore English 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/MERaLiON/sea_audiobench_datasets_TCQ", "isWer": true } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "tcq_sg_streets_30": 0.37635054, "tcq_sg_streets_60": 1.03159974, "tcq_yodas2_en_30": 0.67530896, "tcq_yodas2_en_60": 2.17655697, "average": 3.1686, "tcq_yodas2_en_120": 5.0269179, "tcq_yodas2_en_180": 8.43540334, "tcq_sgpccsc_120": 14.7541, "tcq_sgpccsc_180": 15.641, "tcq_sgpccsc_30": 18.8672, "tcq_sgpccsc_60": 15.5466, "tcq_sg_streets_120": 4.45833333 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "tcq_sg_streets_30": 3.04801921, "tcq_sg_streets_60": 1.60566162, "tcq_yodas2_en_30": 3.55483536, "tcq_yodas2_en_60": 3.66182949, "average": 3.6405, "tcq_yodas2_en_120": 3.48839166, "tcq_yodas2_en_180": 3.668357, "tcq_sgpccsc_120": 1.0, "tcq_sgpccsc_180": 1.0, "tcq_sgpccsc_30": 1.0, "tcq_sgpccsc_60": 1.0, "tcq_sg_streets_120": 6.45673077 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "tcq_sg_streets_30": 2.01320528, "tcq_sg_streets_60": 2.70704411, "tcq_yodas2_en_30": 3.18166105, "tcq_yodas2_en_60": 6.88278483, "average": 6.0463, "tcq_yodas2_en_120": 12.48973755, "tcq_sg_streets_120": 9.00320513 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "tcq_sg_streets_30": 2.69687875, "tcq_sg_streets_60": 3.66951942, "tcq_yodas2_en_30": 5.44481894, "tcq_yodas2_en_60": 11.55359765, "average": 6.2127, "tcq_sgpccsc_120": 40.9742268, "tcq_sgpccsc_180": 58.92857143, "tcq_sgpccsc_30": 9.35985312, "tcq_sgpccsc_60": 19.51515152, "tcq_sg_streets_120": 7.69871795 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "tcq_sg_streets_30": 2.44297719, "tcq_sg_streets_60": 3.37393022, "tcq_yodas2_en_30": 5.31622159, "tcq_yodas2_en_60": 10.79752958, "average": 6.248, "tcq_sg_streets_120": 9.30929487 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "tcq_sg_streets_30": 2.40756303, "tcq_sg_streets_60": 3.27320606, "tcq_yodas2_en_30": 5.43367038, "tcq_yodas2_en_60": 11.40735942, "average": 6.2752, "tcq_sgpccsc_120": 61.20618557, "tcq_sgpccsc_180": 84.98412698, "tcq_sgpccsc_30": 12.84822521, "tcq_sgpccsc_60": 31.81060606, "tcq_sg_streets_120": 8.85416667 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "tcq_sg_streets_30": 2.4057623, "tcq_sg_streets_60": 3.42198815, "tcq_yodas2_en_30": 5.4740299, "tcq_yodas2_en_60": 11.33678846, "average": 6.2969, "tcq_sgpccsc_120": 51.57216495, "tcq_sgpccsc_180": 75.53968254, "tcq_sgpccsc_30": 11.28029376, "tcq_sgpccsc_60": 24.90656566, "tcq_sg_streets_120": 8.84615385 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "tcq_sg_streets_30": 1.94177671, "tcq_sg_streets_60": 2.79262673, "tcq_yodas2_en_30": 4.12522686, "tcq_yodas2_en_60": 8.22717457, "average": 6.4614, "tcq_yodas2_en_120": 15.06275236, "tcq_sg_streets_120": 6.61858974 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "tcq_sg_streets_30": 2.63505402, "tcq_sg_streets_60": 3.52271231, "tcq_yodas2_en_30": 5.32166623, "tcq_yodas2_en_60": 11.05251792, "average": 8.8354, "tcq_yodas2_en_120": 20.68093876, "tcq_sg_streets_120": 9.79967949 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "tcq_sg_streets_30": 2.3967587, "tcq_sg_streets_60": 3.32982225, "tcq_yodas2_en_30": 5.26004667, "tcq_yodas2_en_60": 11.22078259, "average": 8.9014, "tcq_yodas2_en_120": 21.69136945, "tcq_sg_streets_120": 9.50961538 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "tcq_sg_streets_30": 2.07322929, "tcq_sg_streets_60": 3.05661619, "tcq_yodas2_en_30": 5.09739867, "tcq_yodas2_en_60": 10.24911462, "average": 10.9611, "tcq_yodas2_en_120": 19.03255384, "tcq_yodas2_en_180": 29.74629427, "tcq_sg_streets_120": 7.47275641 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "tcq_yodas2_en_30": 5.29720854, "tcq_yodas2_en_60": 11.07324868, "tcq_yodas2_en_120": 21.81140646, "tcq_yodas2_en_180": 33.05890154, "average": 14.2107, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT", "tcq_sg_streets_60": 3.88545095, "tcq_sg_streets_120": 10.13782051 } ], "metric": "wer", "metricInfo": "Word/Character Error Rate over the queried time interval - The lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "TCQ-Mandarin", "title": "Task: TCQ - Mandarin", "taskName": "tcq_mandarin", "metric": "wer", "metricInfo": "Word/Character Error Rate over the queried time interval - The lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "TCQ-Yodas2-ZH-30 [SEA]", "internal": "tcq_yodas2_zh_30", "description": "TCQ YODAS2 Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-ZH-60 [SEA]", "internal": "tcq_yodas2_zh_60", "description": "TCQ YODAS2 Chinese 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-ZH-120 [SEA]", "internal": "tcq_yodas2_zh_120", "description": "TCQ YODAS2 Chinese 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-ZH-180 [SEA]", "internal": "tcq_yodas2_zh_180", "description": "TCQ YODAS2 Chinese 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Sgpccsc-30 [SEA]", "internal": "tcq_sgpccsc_30", "description": "TCQ SGPCCSC Mandarin 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Sgpccsc-60 [SEA]", "internal": "tcq_sgpccsc_60", "description": "TCQ SGPCCSC Mandarin 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Sgpccsc-120 [SEA]", "internal": "tcq_sgpccsc_120", "description": "TCQ SGPCCSC Mandarin 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Sgpccsc-180 [SEA]", "internal": "tcq_sgpccsc_180", "description": "TCQ SGPCCSC Mandarin 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} } ], "data": { "columns": [ { "key": "tcq_yodas2_zh_30", "display": "TCQ-Yodas2-ZH-30 [SEA]", "description": "TCQ YODAS2 Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_zh_60", "display": "TCQ-Yodas2-ZH-60 [SEA]", "description": "TCQ YODAS2 Chinese 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_zh_120", "display": "TCQ-Yodas2-ZH-120 [SEA]", "description": "TCQ YODAS2 Chinese 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_zh_180", "display": "TCQ-Yodas2-ZH-180 [SEA]", "description": "TCQ YODAS2 Chinese 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_sgpccsc_30", "display": "TCQ-Sgpccsc-30 [SEA]", "description": "TCQ SGPCCSC Mandarin 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_sgpccsc_60", "display": "TCQ-Sgpccsc-60 [SEA]", "description": "TCQ SGPCCSC Mandarin 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_sgpccsc_120", "display": "TCQ-Sgpccsc-120 [SEA]", "description": "TCQ SGPCCSC Mandarin 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_sgpccsc_180", "display": "TCQ-Sgpccsc-180 [SEA]", "description": "TCQ SGPCCSC Mandarin 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true } ], "rows": [ { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "tcq_yodas2_zh_30": 3.84671533, "tcq_yodas2_zh_60": 3.98823529, "tcq_yodas2_zh_120": 3.50580495, "tcq_yodas2_zh_180": 3.2474645, "average": 3.7745, "tcq_sgpccsc_30": 3.5362458, "tcq_sgpccsc_60": 4.17035512, "tcq_sgpccsc_120": 4.73768939, "tcq_sgpccsc_180": 3.16360856 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "tcq_yodas2_zh_30": 0.70406674, "tcq_yodas2_zh_60": 2.68686731, "tcq_yodas2_zh_120": 6.10371517, "tcq_yodas2_zh_180": 11.60446247, "average": 6.7658, "tcq_sgpccsc_30": 0.98463754, "tcq_sgpccsc_180": 18.60703364, "tcq_sgpccsc_60": 3.1142563, "tcq_sgpccsc_120": 10.32102273 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "tcq_yodas2_zh_30": 4.6649635, "tcq_yodas2_zh_60": 8.4875513, "tcq_yodas2_zh_120": 15.56965944, "tcq_yodas2_zh_180": 19.81135903, "average": 12.5149, "tcq_sgpccsc_30": 4.5962554, "tcq_sgpccsc_60": 8.91971179, "tcq_sgpccsc_120": 15.24715909, "tcq_sgpccsc_180": 22.82262997 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "tcq_yodas2_zh_30": 5.28561001, "tcq_yodas2_zh_60": 10.0128591, "tcq_yodas2_zh_120": 17.43343653, "tcq_yodas2_zh_180": 29.55780933, "average": 15.2714, "tcq_sgpccsc_30": 4.43326932, "tcq_sgpccsc_60": 10.23571796, "tcq_sgpccsc_120": 18.23390152, "tcq_sgpccsc_180": 26.97859327 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "tcq_yodas2_zh_30": 11.92565172, "tcq_yodas2_zh_60": 21.31751026, "tcq_yodas2_zh_120": 38.13660991, "average": 23.5776, "tcq_sgpccsc_30": 9.93734998, "tcq_sgpccsc_60": 21.2110139, "tcq_sgpccsc_120": 38.9375 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "tcq_yodas2_zh_30": 11.55839416, "tcq_yodas2_zh_60": 20.71422709, "tcq_yodas2_zh_120": 37.18111455, "tcq_yodas2_zh_180": 59.34077079, "average": 27.6094, "tcq_sgpccsc_30": 9.05256841, "tcq_sgpccsc_60": 20.41585178, "tcq_sgpccsc_120": 35.00284091 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "tcq_yodas2_zh_30": 11.06402503, "tcq_yodas2_zh_60": 19.54131327, "tcq_yodas2_zh_120": 33.87848297, "tcq_yodas2_zh_180": 53.65720081, "average": 28.2432, "tcq_sgpccsc_30": 8.67738838, "tcq_sgpccsc_60": 16.1868245, "tcq_sgpccsc_120": 31.59943182, "tcq_sgpccsc_180": 51.34097859 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "tcq_yodas2_zh_30": 13.87142857, "tcq_yodas2_zh_60": 25.95348837, "tcq_yodas2_zh_120": 45.80147059, "average": 28.9643, "tcq_sgpccsc_30": 12.27604417, "tcq_sgpccsc_60": 26.1626351, "tcq_sgpccsc_120": 49.72064394 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "tcq_yodas2_zh_30": 11.9919708, "tcq_yodas2_zh_60": 22.92749658, "tcq_yodas2_zh_120": 38.88622291, "tcq_yodas2_zh_180": 62.12778905, "average": 33.0889, "tcq_sgpccsc_60": 19.5152, "tcq_sgpccsc_180": 58.9286, "tcq_sgpccsc_30": 9.3599, "tcq_sgpccsc_120": 40.9742 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "tcq_sgpccsc_30": 10.8456553, "tcq_sgpccsc_60": 22.01029336, "tcq_sgpccsc_120": 40.76136364, "tcq_sgpccsc_180": 63.89449541, "tcq_yodas2_zh_30": 12.66934307, "tcq_yodas2_zh_60": 22.11409029, "tcq_yodas2_zh_120": 39.61068111, "tcq_yodas2_zh_180": 65.26369168, "average": 34.6462, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "tcq_yodas2_zh_30": 11.85849844, "tcq_yodas2_zh_60": 22.25239398, "tcq_yodas2_zh_120": 38.60681115, "tcq_yodas2_zh_180": 63.50912779, "average": 37.4407, "tcq_sgpccsc_30": 11.2803, "tcq_sgpccsc_180": 75.5397, "tcq_sgpccsc_60": 24.9066, "tcq_sgpccsc_120": 51.5722 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "tcq_yodas2_zh_30": 12.60490094, "tcq_yodas2_zh_60": 22.4254446, "tcq_yodas2_zh_120": 39.69620743, "tcq_yodas2_zh_180": 63.01419878, "average": 41.0737, "tcq_sgpccsc_180": 84.9841, "tcq_sgpccsc_30": 12.8482, "tcq_sgpccsc_60": 31.8106, "tcq_sgpccsc_120": 61.2062 } ], "metric": "wer", "metricInfo": "Word/Character Error Rate over the queried time interval - The lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "TCQ-Indonesian", "title": "Task: TCQ - Indonesian", "taskName": "tcq_indonesian", "metric": "wer", "metricInfo": "Word/Character Error Rate over the queried time interval - The lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "TCQ-Yodas2-ID-30 [SEA]", "internal": "tcq_yodas2_id_30", "description": "TCQ YODAS2 Indonesian 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-ID-60 [SEA]", "internal": "tcq_yodas2_id_60", "description": "TCQ YODAS2 Indonesian 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-ID-120 [SEA]", "internal": "tcq_yodas2_id_120", "description": "TCQ YODAS2 Indonesian 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-ID-180 [SEA]", "internal": "tcq_yodas2_id_180", "description": "TCQ YODAS2 Indonesian 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} } ], "data": { "columns": [ { "key": "tcq_yodas2_id_30", "display": "TCQ-Yodas2-ID-30 [SEA]", "description": "TCQ YODAS2 Indonesian 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_id_60", "display": "TCQ-Yodas2-ID-60 [SEA]", "description": "TCQ YODAS2 Indonesian 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_id_120", "display": "TCQ-Yodas2-ID-120 [SEA]", "description": "TCQ YODAS2 Indonesian 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_id_180", "display": "TCQ-Yodas2-ID-180 [SEA]", "description": "TCQ YODAS2 Indonesian 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true } ], "rows": [ { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "tcq_yodas2_id_30": 3.12356348, "tcq_yodas2_id_60": 3.5603671, "tcq_yodas2_id_120": 3.72439549, "tcq_yodas2_id_180": 3.61392949, "average": 3.5056 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "tcq_yodas2_id_30": 2.23591064, "tcq_yodas2_id_60": 4.21607413, "tcq_yodas2_id_120": 7.59754399, "average": 4.6832 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "tcq_yodas2_id_30": 3.78836076, "tcq_yodas2_id_60": 7.79559832, "average": 5.792 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "tcq_yodas2_id_30": 3.29475039, "tcq_yodas2_id_60": 7.02263209, "tcq_yodas2_id_120": 12.71312824, "average": 7.6768 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "tcq_yodas2_id_30": 3.75039073, "tcq_yodas2_id_60": 7.21144079, "tcq_yodas2_id_120": 13.10039245, "average": 8.0207 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "tcq_yodas2_id_30": 0.66718764, "tcq_yodas2_id_60": 2.28708901, "tcq_yodas2_id_120": 4.97708571, "tcq_yodas2_id_180": 31.82481752, "average": 9.939 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "tcq_yodas2_id_30": 3.45849039, "tcq_yodas2_id_60": 6.63298583, "tcq_yodas2_id_120": 12.01810356, "tcq_yodas2_id_180": 18.68157065, "average": 10.1978 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "tcq_yodas2_id_30": 4.5154914, "tcq_yodas2_id_60": 9.50788559, "tcq_yodas2_id_120": 17.02531966, "average": 10.3496 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "tcq_yodas2_id_30": 3.76142319, "tcq_yodas2_id_60": 7.90947162, "tcq_yodas2_id_120": 14.72009115, "tcq_yodas2_id_180": 22.48953855, "average": 12.2201 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "tcq_yodas2_id_30": 3.91808403, "tcq_yodas2_id_60": 8.01826606, "tcq_yodas2_id_120": 14.85504494, "tcq_yodas2_id_180": 22.49441101, "average": 12.3215 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "tcq_yodas2_id_30": 3.8119886, "tcq_yodas2_id_60": 8.0814399, "tcq_yodas2_id_120": 15.22559818, "tcq_yodas2_id_180": 22.7033534, "average": 12.4556, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "tcq_yodas2_id_30": 4.2600901, "tcq_yodas2_id_60": 8.41201105, "tcq_yodas2_id_120": 15.07532599, "tcq_yodas2_id_180": 22.92433362, "average": 12.6679 } ], "metric": "wer", "metricInfo": "Word/Character Error Rate over the queried time interval - The lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "TCQ-Thai", "title": "Task: TCQ - Thai", "taskName": "tcq_thai", "metric": "wer", "metricInfo": "Word/Character Error Rate over the queried time interval - The lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "TCQ-Yodas2-TH-30 [SEA]", "internal": "tcq_yodas2_th_30", "description": "TCQ YODAS2 Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-TH-60 [SEA]", "internal": "tcq_yodas2_th_60", "description": "TCQ YODAS2 Thai 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-TH-180 [SEA]", "internal": "tcq_yodas2_th_180", "description": "TCQ YODAS2 Thai 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-TH-120 [SEA]", "internal": "tcq_yodas2_th_120", "description": "TCQ YODAS2 Thai 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} } ], "data": { "columns": [ { "key": "tcq_yodas2_th_30", "display": "TCQ-Yodas2-TH-30 [SEA]", "description": "TCQ YODAS2 Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_th_60", "display": "TCQ-Yodas2-TH-60 [SEA]", "description": "TCQ YODAS2 Thai 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_th_180", "display": "TCQ-Yodas2-TH-180 [SEA]", "description": "TCQ YODAS2 Thai 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_th_120", "display": "TCQ-Yodas2-TH-120 [SEA]", "description": "TCQ YODAS2 Thai 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true } ], "rows": [ { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "tcq_yodas2_th_30": 2.51196608, "tcq_yodas2_th_60": 2.8209015, "tcq_yodas2_th_180": 2.42277504, "average": 2.6072, "tcq_yodas2_th_120": 2.67313489 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "tcq_yodas2_th_30": 0.65502513, "tcq_yodas2_th_60": 2.27285331, "tcq_yodas2_th_180": 25.94059406, "average": 8.4192, "tcq_yodas2_th_120": 4.80844009 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "tcq_yodas2_th_30": 9.05423995, "average": 9.0542 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "average": 11.1639, "tcq_yodas2_th_120": 11.16390354 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "tcq_yodas2_th_30": 4.07889447, "tcq_yodas2_th_60": 8.30249253, "tcq_yodas2_th_180": 23.07024079, "average": 11.8172 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "tcq_yodas2_th_30": 6.98671482, "tcq_yodas2_th_60": 12.45195181, "average": 13.4871, "tcq_yodas2_th_120": 21.02260739 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "tcq_yodas2_th_30": 7.07091709, "tcq_yodas2_th_60": 12.88400037, "average": 14.1681, "tcq_yodas2_th_120": 22.54951017 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "tcq_yodas2_th_30": 5.60772613, "tcq_yodas2_th_60": 9.94059833, "tcq_yodas2_th_180": 27.08833311, "average": 14.9669, "tcq_yodas2_th_120": 17.23078372 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "tcq_yodas2_th_30": 7.2098304, "tcq_yodas2_th_60": 13.63373078, "tcq_yodas2_th_180": 35.7033391, "average": 18.849 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "tcq_yodas2_th_30": 7.39321608, "tcq_yodas2_th_60": 13.66977848, "tcq_yodas2_th_180": 35.65478249, "average": 18.9059 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "tcq_yodas2_th_30": 7.15081658, "tcq_yodas2_th_60": 13.44347906, "tcq_yodas2_th_180": 36.16815219, "average": 18.9208 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "tcq_yodas2_th_30": 7.13809673, "tcq_yodas2_th_60": 13.74791262, "tcq_yodas2_th_120": 24.41914092, "tcq_yodas2_th_180": 36.50392444, "average": 20.4523, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" } ], "metric": "wer", "metricInfo": "Word/Character Error Rate over the queried time interval - The lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "TCQ-Vietnamese", "title": "Task: TCQ - Vietnamese", "taskName": "tcq_vietnamese", "metric": "wer", "metricInfo": "Word/Character Error Rate over the queried time interval - The lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true, "datasets": [ { "display": "TCQ-Yodas2-VI-30 [SEA]", "internal": "tcq_yodas2_vi_30", "description": "TCQ YODAS2 Vietnamese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-VI-60 [SEA]", "internal": "tcq_yodas2_vi_60", "description": "TCQ YODAS2 Vietnamese 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-VI-120 [SEA]", "internal": "tcq_yodas2_vi_120", "description": "TCQ YODAS2 Vietnamese 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} }, { "display": "TCQ-Yodas2-VI-180 [SEA]", "internal": "tcq_yodas2_vi_180", "description": "TCQ YODAS2 Vietnamese 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "stats": {} } ], "data": { "columns": [ { "key": "tcq_yodas2_vi_30", "display": "TCQ-Yodas2-VI-30 [SEA]", "description": "TCQ YODAS2 Vietnamese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_vi_60", "display": "TCQ-Yodas2-VI-60 [SEA]", "description": "TCQ YODAS2 Vietnamese 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_vi_120", "display": "TCQ-Yodas2-VI-120 [SEA]", "description": "TCQ YODAS2 Vietnamese 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true }, { "key": "tcq_yodas2_vi_180", "display": "TCQ-Yodas2-VI-180 [SEA]", "description": "TCQ YODAS2 Vietnamese 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TCQ", "isWer": true } ], "rows": [ { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "tcq_yodas2_vi_30": 4.06771894, "tcq_yodas2_vi_60": 3.39093419, "tcq_yodas2_vi_120": 3.56075608, "tcq_yodas2_vi_180": 2.78409091, "average": 3.4509 }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "tcq_yodas2_vi_30": 0.77002716, "tcq_yodas2_vi_60": 2.30234699, "tcq_yodas2_vi_120": 5.77227723, "tcq_yodas2_vi_180": 9.10416667, "average": 4.4872 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "tcq_yodas2_vi_30": 2.65767142, "tcq_yodas2_vi_60": 4.36838472, "tcq_yodas2_vi_120": 7.85823582, "tcq_yodas2_vi_180": 12.33333333, "average": 6.8044 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "tcq_yodas2_vi_30": 2.96096402, "tcq_yodas2_vi_60": 4.56258629, "tcq_yodas2_vi_120": 9.35778578, "tcq_yodas2_vi_180": 12.89962121, "average": 7.4452 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "tcq_yodas2_vi_30": 5.49389002, "tcq_yodas2_vi_60": 10.51978831, "average": 8.0068 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "tcq_yodas2_vi_30": 4.75797692, "tcq_yodas2_vi_60": 9.3216751, "tcq_yodas2_vi_120": 15.8870387, "average": 9.9889 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "tcq_yodas2_vi_30": 5.24592668, "tcq_yodas2_vi_60": 9.4176254, "tcq_yodas2_vi_120": 16.35643564, "average": 10.34 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "tcq_yodas2_vi_30": 4.18041412, "tcq_yodas2_vi_60": 7.28969167, "tcq_yodas2_vi_120": 12.61386139, "tcq_yodas2_vi_180": 19.20170455, "average": 10.8214 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "tcq_yodas2_vi_30": 5.41259335, "tcq_yodas2_vi_60": 10.45582145, "tcq_yodas2_vi_120": 19.72727273, "tcq_yodas2_vi_180": 26.24621212, "average": 15.4605, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "tcq_yodas2_vi_30": 5.45230821, "tcq_yodas2_vi_60": 10.67234238, "tcq_yodas2_vi_120": 19.69981998, "tcq_yodas2_vi_180": 26.52083333, "average": 15.5863 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "tcq_yodas2_vi_30": 5.47318398, "tcq_yodas2_vi_60": 11.8633226, "tcq_yodas2_vi_120": 19.61656166, "tcq_yodas2_vi_180": 26.11363636, "average": 15.7667 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "tcq_yodas2_vi_30": 6.29972845, "tcq_yodas2_vi_60": 11.5204786, "tcq_yodas2_vi_120": 19.7749775, "tcq_yodas2_vi_180": 27.74715909, "average": 16.3356 } ], "metric": "wer", "metricInfo": "Word/Character Error Rate over the queried time interval - The lower, the better. Note: an error rate above 100% is possible because insertions are unbounded — it usually means the model transcribed the audio in a different language or script than the reference, which happens when the language is outside the model's supported set. Such cells are genuine measurements, not evaluation failures.", "ascending": true } }, { "key": "TLoc-English", "title": "Task: TLoc - English", "taskName": "tloc_english", "metric": "F1", "metricInfo": "F1 of Coverage and Purity for predicted vs gold time intervals - The higher, the better.", "ascending": false, "datasets": [ { "display": "TLOC-SG-Streets-30 [SEA]", "internal": "tloc_sg_streets_30", "description": "TLoc SG Streets Singapore English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} }, { "display": "TLOC-Yodas2-EN-30 [SEA]", "internal": "tloc_yodas2_en_30", "description": "TLoc YODAS2 English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} } ], "data": { "columns": [ { "key": "tloc_sg_streets_30", "display": "TLOC-SG-Streets-30 [SEA]", "description": "TLoc SG Streets Singapore English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false }, { "key": "tloc_yodas2_en_30", "display": "TLOC-Yodas2-EN-30 [SEA]", "description": "TLoc YODAS2 English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "tloc_sg_streets_30": 60.63157895, "tloc_yodas2_en_30": 64.26, "average": 62.4458, "tloc_sgpccsc_long_30": 62.5390625 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "tloc_sg_streets_30": 33.47368421, "tloc_yodas2_en_30": 17.06, "average": 25.2668, "tloc_sgpccsc_long_30": 8.8671875 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "tloc_sg_streets_30": 24.42105263, "tloc_yodas2_en_30": 15.18, "average": 19.8005, "tloc_sgpccsc_long_30": 12.890625 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "tloc_sg_streets_30": 24.0, "tloc_yodas2_en_30": 13.86, "average": 18.93, "tloc_sgpccsc_long_30": 7.4609375 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "tloc_sg_streets_30": 16.42105263, "tloc_yodas2_en_30": 10.2, "average": 13.3105, "tloc_sgpccsc_long_30": 6.1328125 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "tloc_sg_streets_30": 13.47368421, "tloc_yodas2_en_30": 7.08, "average": 10.2768, "tloc_sgpccsc_long_30": 2.9296875 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "tloc_sg_streets_30": 0.5591, "tloc_yodas2_en_30": 0.4198, "average": 0.4895 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "tloc_sg_streets_30": 0.4878, "tloc_yodas2_en_30": 0.3426, "average": 0.4152 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "tloc_sg_streets_30": 0.4433, "tloc_yodas2_en_30": 0.3448, "average": 0.3941 }, { "model": "MERaLiON-3-10B", "tloc_sg_streets_30": 0.42540748, "tloc_yodas2_en_30": 0.29796445, "average": 0.3617, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "tloc_sg_streets_30": 0.3359, "tloc_yodas2_en_30": 0.2293, "average": 0.2826 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "tloc_sg_streets_30": 0.3448, "tloc_yodas2_en_30": 0.1925, "average": 0.2686 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "tloc_sg_streets_30": 0.2118, "tloc_yodas2_en_30": 0.1138, "average": 0.1628 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "tloc_sg_streets_30": 0.0, "tloc_yodas2_en_30": 0.0, "average": 0.0 } ], "metric": "F1", "metricInfo": "F1 of Coverage and Purity for predicted vs gold time intervals - The higher, the better.", "ascending": false } }, { "key": "TLoc-Mandarin", "title": "Task: TLoc - Mandarin", "taskName": "tloc_mandarin", "metric": "F1", "metricInfo": "F1 of Coverage and Purity for predicted vs gold time intervals - The higher, the better.", "ascending": false, "datasets": [ { "display": "TLOC-Yodas2-ZH-30 [SEA]", "internal": "tloc_yodas2_zh_30", "description": "TLoc YODAS2 Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} }, { "display": "TLOC-Yodas2-ZH-60 [SEA]", "internal": "tloc_yodas2_zh_60", "description": "TLoc YODAS2 Chinese 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} }, { "display": "TLOC-Yodas2-ZH-120 [SEA]", "internal": "tloc_yodas2_zh_120", "description": "TLoc YODAS2 Chinese 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} }, { "display": "TLOC-Yodas2-ZH-180 [SEA]", "internal": "tloc_yodas2_zh_180", "description": "TLoc YODAS2 Chinese 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} }, { "display": "TLOC-Sgpccsc-LONG-30 [SEA]", "internal": "tloc_sgpccsc_long_30", "description": "TLoc SGPCCSC (long) Mandarin 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} } ], "data": { "columns": [ { "key": "tloc_yodas2_zh_30", "display": "TLOC-Yodas2-ZH-30 [SEA]", "description": "TLoc YODAS2 Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false }, { "key": "tloc_yodas2_zh_60", "display": "TLOC-Yodas2-ZH-60 [SEA]", "description": "TLoc YODAS2 Chinese 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false }, { "key": "tloc_yodas2_zh_120", "display": "TLOC-Yodas2-ZH-120 [SEA]", "description": "TLoc YODAS2 Chinese 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false }, { "key": "tloc_yodas2_zh_180", "display": "TLOC-Yodas2-ZH-180 [SEA]", "description": "TLoc YODAS2 Chinese 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false }, { "key": "tloc_sgpccsc_long_30", "display": "TLOC-Sgpccsc-LONG-30 [SEA]", "description": "TLoc SGPCCSC (long) Mandarin 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "tloc_yodas2_zh_30": 68.68, "tloc_yodas2_zh_60": 34.85411141, "tloc_yodas2_zh_120": 13.72693727, "tloc_yodas2_zh_180": 15.10204082, "average": 38.9804, "tloc_sgpccsc_long_30": 62.5391 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "tloc_yodas2_zh_30": 14.12, "tloc_yodas2_zh_60": 11.85676393, "tloc_yodas2_zh_120": 8.26568266, "tloc_yodas2_zh_180": 13.87755102, "average": 12.2021, "tloc_sgpccsc_long_30": 12.8906 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "tloc_yodas2_zh_30": 7.44, "tloc_yodas2_zh_60": 9.44297082, "tloc_yodas2_zh_120": 8.04428044, "tloc_yodas2_zh_180": 9.3877551, "tloc_sgpccsc_long_30": 8.8672, "average": 8.6364 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "tloc_yodas2_zh_30": 8.6, "tloc_yodas2_zh_60": 7.32095491, "tloc_yodas2_zh_120": 4.50184502, "tloc_yodas2_zh_180": 5.71428571, "average": 6.7196, "tloc_sgpccsc_long_30": 7.4609 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "tloc_yodas2_zh_30": 5.7, "tloc_yodas2_zh_60": 3.79310345, "tloc_yodas2_zh_120": 2.73062731, "tloc_yodas2_zh_180": 2.44897959, "average": 4.1611, "tloc_sgpccsc_long_30": 6.1328 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "tloc_yodas2_zh_30": 3.04, "tloc_yodas2_zh_60": 2.12201592, "tloc_yodas2_zh_120": 0.14760148, "tloc_yodas2_zh_180": 0.40816327, "average": 1.7295, "tloc_sgpccsc_long_30": 2.9297 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "tloc_yodas2_zh_30": 0.2105, "tloc_yodas2_zh_60": 0.1286, "tloc_yodas2_zh_120": 0.089, "tloc_yodas2_zh_180": 0.0592, "average": 0.1218 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "tloc_yodas2_zh_30": 0.1891, "tloc_yodas2_zh_60": 0.1264, "tloc_yodas2_zh_120": 0.0812, "tloc_yodas2_zh_180": 0.076, "average": 0.1182 }, { "model": "MERaLiON-3-10B", "tloc_sgpccsc_long_30": 0.16724723, "tloc_yodas2_zh_30": 0.18338148, "tloc_yodas2_zh_60": 0.11815699, "tloc_yodas2_zh_120": 0.05523703, "tloc_yodas2_zh_180": 0.0502944, "average": 0.1149, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "tloc_yodas2_zh_30": 0.2626, "tloc_yodas2_zh_60": 0.0795, "tloc_yodas2_zh_120": 0.0417, "tloc_yodas2_zh_180": 0.0, "average": 0.096 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "tloc_yodas2_zh_30": 0.1246, "tloc_yodas2_zh_60": 0.0917, "tloc_yodas2_zh_120": 0.0509, "tloc_yodas2_zh_180": 0.0584, "average": 0.0814 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "tloc_yodas2_zh_30": 0.1071, "tloc_yodas2_zh_60": 0.0577, "tloc_yodas2_zh_120": 0.0333, "tloc_yodas2_zh_180": 0.0112, "average": 0.0523 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "tloc_yodas2_zh_30": 0.0609, "tloc_yodas2_zh_60": 0.0537, "tloc_yodas2_zh_120": 0.0336, "tloc_yodas2_zh_180": 0.0212, "average": 0.0423 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "tloc_yodas2_zh_30": 0.0, "tloc_yodas2_zh_60": 0.0, "tloc_yodas2_zh_120": 0.0, "tloc_yodas2_zh_180": 0.0, "average": 0.0 } ], "metric": "F1", "metricInfo": "F1 of Coverage and Purity for predicted vs gold time intervals - The higher, the better.", "ascending": false } }, { "key": "TLoc-Indonesian", "title": "Task: TLoc - Indonesian", "taskName": "tloc_indonesian", "metric": "F1", "metricInfo": "F1 of Coverage and Purity for predicted vs gold time intervals - The higher, the better.", "ascending": false, "datasets": [ { "display": "TLOC-Yodas2-ID-30 [SEA]", "internal": "tloc_yodas2_id_30", "description": "TLoc YODAS2 Indonesian 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} } ], "data": { "columns": [ { "key": "tloc_yodas2_id_30", "display": "TLOC-Yodas2-ID-30 [SEA]", "description": "TLoc YODAS2 Indonesian 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "tloc_yodas2_id_30": 67.52, "average": 67.52 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "tloc_yodas2_id_30": 23.2, "average": 23.2 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "tloc_yodas2_id_30": 21.24, "average": 21.24 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "tloc_yodas2_id_30": 18.02, "average": 18.02 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "tloc_yodas2_id_30": 12.94, "average": 12.94 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "tloc_yodas2_id_30": 9.26, "average": 9.26 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "tloc_yodas2_id_30": 0.4386, "average": 0.4386 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "tloc_yodas2_id_30": 0.4054, "average": 0.4054 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "tloc_yodas2_id_30": 0.384, "average": 0.384 }, { "model": "MERaLiON-3-10B", "tloc_yodas2_id_30": 0.35208709, "average": 0.3521, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "tloc_yodas2_id_30": 0.2699, "average": 0.2699 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "tloc_yodas2_id_30": 0.2283, "average": 0.2283 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "tloc_yodas2_id_30": 0.1849, "average": 0.1849 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "tloc_yodas2_id_30": 0.0, "average": 0.0 } ], "metric": "F1", "metricInfo": "F1 of Coverage and Purity for predicted vs gold time intervals - The higher, the better.", "ascending": false } }, { "key": "TLoc-Thai", "title": "Task: TLoc - Thai", "taskName": "tloc_thai", "metric": "F1", "metricInfo": "F1 of Coverage and Purity for predicted vs gold time intervals - The higher, the better.", "ascending": false, "datasets": [ { "display": "TLOC-Yodas2-TH-30 [SEA]", "internal": "tloc_yodas2_th_30", "description": "TLoc YODAS2 Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} }, { "display": "TLOC-Yodas2-TH-60 [SEA]", "internal": "tloc_yodas2_th_60", "description": "TLoc YODAS2 Thai 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} } ], "data": { "columns": [ { "key": "tloc_yodas2_th_30", "display": "TLOC-Yodas2-TH-30 [SEA]", "description": "TLoc YODAS2 Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false }, { "key": "tloc_yodas2_th_60", "display": "TLOC-Yodas2-TH-60 [SEA]", "description": "TLoc YODAS2 Thai 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "tloc_yodas2_th_30": 63.14, "tloc_yodas2_th_60": 34.3, "average": 48.72 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "tloc_yodas2_th_30": 16.2, "tloc_yodas2_th_60": 11.08, "average": 13.64 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "tloc_yodas2_th_30": 14.8, "tloc_yodas2_th_60": 11.18, "average": 12.99 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "tloc_yodas2_th_30": 12.92, "tloc_yodas2_th_60": 7.24, "average": 10.08 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "tloc_yodas2_th_30": 9.32, "tloc_yodas2_th_60": 6.5, "average": 7.91 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "tloc_yodas2_th_30": 6.28, "tloc_yodas2_th_60": 4.42, "average": 5.35 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "tloc_yodas2_th_30": 0.3249, "tloc_yodas2_th_60": 0.1781, "average": 0.2515 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "tloc_yodas2_th_30": 0.269, "tloc_yodas2_th_60": 0.1388, "average": 0.2039 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "tloc_yodas2_th_30": 0.1652, "tloc_yodas2_th_60": 0.1113, "average": 0.1383 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "tloc_yodas2_th_30": 0.1866, "tloc_yodas2_th_60": 0.0696, "average": 0.1281 }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "tloc_yodas2_th_30": 0.1588, "tloc_yodas2_th_60": 0.0671, "average": 0.1129 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "tloc_yodas2_th_30": 0.107, "tloc_yodas2_th_60": 0.0961, "average": 0.1016 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "tloc_yodas2_th_30": 0.0, "tloc_yodas2_th_60": 0.0, "average": 0.0 } ], "metric": "F1", "metricInfo": "F1 of Coverage and Purity for predicted vs gold time intervals - The higher, the better.", "ascending": false } }, { "key": "TLoc-Vietnamese", "title": "Task: TLoc - Vietnamese", "taskName": "tloc_vietnamese", "metric": "F1", "metricInfo": "F1 of Coverage and Purity for predicted vs gold time intervals - The higher, the better.", "ascending": false, "datasets": [ { "display": "TLOC-Yodas2-VI-30 [SEA]", "internal": "tloc_yodas2_vi_30", "description": "TLoc YODAS2 Vietnamese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} }, { "display": "TLOC-Yodas2-VI-60 [SEA]", "internal": "tloc_yodas2_vi_60", "description": "TLoc YODAS2 Vietnamese 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} }, { "display": "TLOC-Yodas2-VI-120 [SEA]", "internal": "tloc_yodas2_vi_120", "description": "TLoc YODAS2 Vietnamese 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} }, { "display": "TLOC-Yodas2-VI-180 [SEA]", "internal": "tloc_yodas2_vi_180", "description": "TLoc YODAS2 Vietnamese 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "stats": {} } ], "data": { "columns": [ { "key": "tloc_yodas2_vi_30", "display": "TLOC-Yodas2-VI-30 [SEA]", "description": "TLoc YODAS2 Vietnamese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false }, { "key": "tloc_yodas2_vi_60", "display": "TLOC-Yodas2-VI-60 [SEA]", "description": "TLoc YODAS2 Vietnamese 60s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false }, { "key": "tloc_yodas2_vi_120", "display": "TLOC-Yodas2-VI-120 [SEA]", "description": "TLoc YODAS2 Vietnamese 120s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false }, { "key": "tloc_yodas2_vi_180", "display": "TLOC-Yodas2-VI-180 [SEA]", "description": "TLoc YODAS2 Vietnamese 180s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_TLoc", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "tloc_yodas2_vi_30": 64.70347648, "tloc_yodas2_vi_60": 31.20234604, "tloc_yodas2_vi_120": 13.25153374, "tloc_yodas2_vi_180": 8.38709677, "average": 29.3861 }, { "model": "gemma-4-E4B-it", "modelLink": "https://huggingface.co/google/gemma-4-E4B-it", "tloc_yodas2_vi_30": 17.83231084, "tloc_yodas2_vi_60": 14.25219941, "tloc_yodas2_vi_120": 11.04294479, "tloc_yodas2_vi_180": 16.12903226, "average": 14.8141 }, { "model": "gemma-3n-e4b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e4b-it", "tloc_yodas2_vi_30": 17.42331288, "tloc_yodas2_vi_60": 14.25219941, "tloc_yodas2_vi_120": 10.67484663, "tloc_yodas2_vi_180": 8.70967742, "average": 12.765 }, { "model": "gemma-3n-e2b-it", "modelLink": "https://huggingface.co/google/gemma-3n-e2b-it", "tloc_yodas2_vi_30": 16.07361963, "tloc_yodas2_vi_60": 11.14369501, "tloc_yodas2_vi_120": 7.36196319, "tloc_yodas2_vi_180": 5.80645161, "average": 10.0964 }, { "model": "Voxtral-Small-24B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Small-24B-2507", "tloc_yodas2_vi_30": 10.34764826, "tloc_yodas2_vi_60": 6.68621701, "tloc_yodas2_vi_120": 4.78527607, "tloc_yodas2_vi_180": 2.90322581, "average": 6.1806 }, { "model": "Voxtral-Mini-3B-2507", "modelLink": "https://huggingface.co/mistralai/Voxtral-Mini-3B-2507", "tloc_yodas2_vi_30": 8.05725971, "tloc_yodas2_vi_60": 5.27859238, "tloc_yodas2_vi_120": 2.20858896, "tloc_yodas2_vi_180": 1.93548387, "average": 4.37 }, { "model": "Qwen2.5-Omni-7B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-7B", "tloc_yodas2_vi_30": 0.376, "tloc_yodas2_vi_60": 0.2458, "tloc_yodas2_vi_120": 0.1286, "tloc_yodas2_vi_180": 0.125, "average": 0.2188 }, { "model": "Qwen2.5-Omni-3B", "modelLink": "https://huggingface.co/Qwen/Qwen2.5-Omni-3B", "tloc_yodas2_vi_30": 0.3581, "tloc_yodas2_vi_60": 0.184, "tloc_yodas2_vi_120": 0.0957, "tloc_yodas2_vi_180": 0.1004, "average": 0.1846 }, { "model": "MERaLiON-3-10B", "tloc_yodas2_vi_60": 0.18835564, "tloc_yodas2_vi_120": 0.08766746, "tloc_yodas2_vi_180": 0.07501373, "average": 0.117, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "Qwen2-Audio-7B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen2-Audio-7B-Instruct", "tloc_yodas2_vi_30": 0.3442, "tloc_yodas2_vi_60": 0.1001, "tloc_yodas2_vi_120": 0.0119, "tloc_yodas2_vi_180": 0.0, "average": 0.1141 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "tloc_yodas2_vi_30": 0.2197, "tloc_yodas2_vi_60": 0.135, "tloc_yodas2_vi_120": 0.0574, "tloc_yodas2_vi_180": 0.0394, "average": 0.1129 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "tloc_yodas2_vi_30": 0.1777, "tloc_yodas2_vi_60": 0.1042, "tloc_yodas2_vi_120": 0.0655, "tloc_yodas2_vi_180": 0.0239, "average": 0.0928 }, { "model": "SeaLLMs-Audio-7B", "modelLink": "https://huggingface.co/SeaLLMs/SeaLLMs-Audio-7B", "tloc_yodas2_vi_30": 0.0985, "tloc_yodas2_vi_60": 0.1138, "tloc_yodas2_vi_120": 0.0721, "tloc_yodas2_vi_180": 0.0833, "average": 0.0919 }, { "model": "Phi-4-multimodal-instruct", "modelLink": "https://huggingface.co/microsoft/Phi-4-multimodal-instruct", "tloc_yodas2_vi_30": 0.0, "tloc_yodas2_vi_60": 0.0, "tloc_yodas2_vi_120": 0.0, "tloc_yodas2_vi_180": 0.0, "average": 0.0 } ], "metric": "F1", "metricInfo": "F1 of Coverage and Purity for predicted vs gold time intervals - The higher, the better.", "ascending": false } }, { "key": "Age Prediction", "title": "Task: Age Prediction", "taskName": "agep", "metric": "accuracy", "metricInfo": "Accuracy - The higher, the better.", "ascending": false, "datasets": [ { "display": "AGE-Cv21-EN-30 [SEA]", "internal": "age_cv21_en_30", "description": "Age CV21 English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "AGE-Cv21-TA-30 [SEA]", "internal": "age_cv21_ta_30", "description": "Age CV21 Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "AGE-Cv21-TH-30 [SEA]", "internal": "age_cv21_th_30", "description": "Age CV21 Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "AGE-Cv21-VI-30 [SEA]", "internal": "age_cv21_vi_30", "description": "Age CV21 Vietnamese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} }, { "display": "AGE-Cv21-ZH-30 [SEA]", "internal": "age_cv21_zh_30", "description": "Age CV21 Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "stats": {} } ], "data": { "columns": [ { "key": "age_cv21_en_30", "display": "AGE-Cv21-EN-30 [SEA]", "description": "Age CV21 English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "age_cv21_ta_30", "display": "AGE-Cv21-TA-30 [SEA]", "description": "Age CV21 Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "age_cv21_th_30", "display": "AGE-Cv21-TH-30 [SEA]", "description": "Age CV21 Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "age_cv21_vi_30", "display": "AGE-Cv21-VI-30 [SEA]", "description": "Age CV21 Vietnamese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false }, { "key": "age_cv21_zh_30", "display": "AGE-Cv21-ZH-30 [SEA]", "description": "Age CV21 Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_PQA", "isWer": false } ], "rows": [ { "model": "MERaLiON-3-10B", "age_cv21_en_30": 66.0, "age_cv21_ta_30": 79.8, "age_cv21_th_30": 84.25806452, "age_cv21_vi_30": 91.95678271, "age_cv21_zh_30": 77.2, "average": 79.843, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B" }, { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "age_cv21_en_30": 64.1, "age_cv21_ta_30": 73.1, "age_cv21_th_30": 78.1935, "age_cv21_vi_30": 84.6339, "age_cv21_zh_30": 74.9, "average": 74.9855 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "age_cv21_en_30": 63.8, "age_cv21_ta_30": 69.5, "age_cv21_th_30": 70.7097, "age_cv21_vi_30": 63.2053, "age_cv21_zh_30": 76.3, "average": 68.703 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "age_cv21_en_30": 64.35, "age_cv21_ta_30": 67.6, "age_cv21_th_30": 59.871, "age_cv21_vi_30": 72.1489, "age_cv21_zh_30": 73.2, "average": 67.434 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "age_cv21_en_30": 61.75, "age_cv21_ta_30": 68.75, "age_cv21_th_30": 66.96774194, "age_cv21_vi_30": 73.16926771, "age_cv21_zh_30": 66.05, "average": 67.3374, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT" }, { "model": "MERaLiON-3-3B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR", "age_cv21_ta_30": 57.05, "age_cv21_th_30": 40.25806452, "age_cv21_vi_30": 38.23529412, "average": 42.3887, "age_cv21_en_30": 43.55, "age_cv21_zh_30": 32.85 } ], "metric": "accuracy", "metricInfo": "Accuracy - The higher, the better.", "ascending": false } }, { "key": "Speaker Recognition", "title": "Task: Speaker Recognition", "taskName": "spr", "metric": "accuracy", "metricInfo": "Accuracy - The higher, the better.", "ascending": false, "datasets": [ { "display": "SPR-Emota-TA-30 [SEA]", "internal": "spr_emota_ta_30", "description": "Speaker Recognition EMOTA Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "stats": {} }, { "display": "SPR-ESD-EN-30 [SEA]", "internal": "spr_esd_en_30", "description": "Speaker Recognition ESD English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "stats": {} }, { "display": "SPR-ESD-ZH-30 [SEA]", "internal": "spr_esd_zh_30", "description": "Speaker Recognition ESD Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "stats": {} }, { "display": "SPR-MIG-MY-30 [SEA]", "internal": "spr_mig_my_30", "description": "Speaker Recognition MIG Burmese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "stats": {} }, { "display": "SPR-Smaldusc-30 [SEA]", "internal": "spr_smaldusc_30", "description": "Speaker Recognition SMALDUSC Singapore English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "stats": {} }, { "display": "SPR-THAI-Elderly-TH-30 [SEA]", "internal": "spr_thai_elderly_th_30", "description": "Speaker Recognition Thai Elderly Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "stats": {} }, { "display": "SPR-THAI-SER-30 [SEA]", "internal": "spr_thai_ser_30", "description": "Speaker Recognition Thai SER Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "stats": {} }, { "display": "SPR-Voxvietnam-VI-30 [SEA]", "internal": "spr_voxvietnam_vi_30", "description": "Speaker Recognition VoxVietnam Vietnamese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "stats": {} }, { "display": "SPR-Sfdusc-30 [SEA]", "internal": "spr_sfdusc_30", "description": "Speaker Recognition SFDUSC Tagalog 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/MERaLiON/sea_audiobench_datasets_SpR", "stats": {} } ], "data": { "columns": [ { "key": "spr_emota_ta_30", "display": "SPR-Emota-TA-30 [SEA]", "description": "Speaker Recognition EMOTA Tamil 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "isWer": false }, { "key": "spr_esd_en_30", "display": "SPR-ESD-EN-30 [SEA]", "description": "Speaker Recognition ESD English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "isWer": false }, { "key": "spr_esd_zh_30", "display": "SPR-ESD-ZH-30 [SEA]", "description": "Speaker Recognition ESD Chinese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "isWer": false }, { "key": "spr_mig_my_30", "display": "SPR-MIG-MY-30 [SEA]", "description": "Speaker Recognition MIG Burmese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "isWer": false }, { "key": "spr_smaldusc_30", "display": "SPR-Smaldusc-30 [SEA]", "description": "Speaker Recognition SMALDUSC Singapore English 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "isWer": false }, { "key": "spr_thai_elderly_th_30", "display": "SPR-THAI-Elderly-TH-30 [SEA]", "description": "Speaker Recognition Thai Elderly Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "isWer": false }, { "key": "spr_thai_ser_30", "display": "SPR-THAI-SER-30 [SEA]", "description": "Speaker Recognition Thai SER Thai 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "isWer": false }, { "key": "spr_voxvietnam_vi_30", "display": "SPR-Voxvietnam-VI-30 [SEA]", "description": "Speaker Recognition VoxVietnam Vietnamese 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/zxl/sea_audiobench_datasets_SpR", "isWer": false }, { "key": "spr_sfdusc_30", "display": "SPR-Sfdusc-30 [SEA]", "description": "Speaker Recognition SFDUSC Tagalog 30s (AudioBench-SEA dataset)", "hfLink": "https://huggingface.co/datasets/MERaLiON/sea_audiobench_datasets_SpR", "isWer": false } ], "rows": [ { "model": "Qwen3-Omni-30B-A3B-Instruct", "modelLink": "https://huggingface.co/Qwen/Qwen3-Omni-30B-A3B-Instruct", "spr_emota_ta_30": 74.8932, "spr_esd_en_30": 62.4, "spr_esd_zh_30": 58.6, "spr_mig_my_30": 66.5, "spr_smaldusc_30": 70.5, "spr_thai_elderly_th_30": 73.0208, "spr_thai_ser_30": 61.7615, "spr_voxvietnam_vi_30": 76.7, "average": 69.4084, "spr_sfdusc_30": 80.3 }, { "model": "MERaLiON-3-10B", "spr_emota_ta_30": 72.43589744, "spr_esd_en_30": 62.3, "spr_esd_zh_30": 58.6, "spr_mig_my_30": 70.8, "spr_smaldusc_30": 68.1, "spr_thai_elderly_th_30": 62.5, "spr_thai_ser_30": 64.87647691, "spr_voxvietnam_vi_30": 69.4, "average": 67.2458, "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-10B", "spr_sfdusc_30": 76.2 }, { "model": "MERaLiON-2-10B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-10B", "spr_emota_ta_30": 52.5641, "spr_esd_en_30": 50.6, "spr_esd_zh_30": 55.2, "spr_mig_my_30": 48.3, "spr_smaldusc_30": 61.0, "spr_thai_elderly_th_30": 52.5, "spr_thai_ser_30": 51.0204, "spr_voxvietnam_vi_30": 50.8, "average": 53.3483, "spr_sfdusc_30": 58.15 }, { "model": "MERaLiON-2-3B", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-2-3B", "spr_emota_ta_30": 44.4444, "spr_esd_en_30": 41.7, "spr_esd_zh_30": 45.3, "spr_mig_my_30": 36.6, "spr_smaldusc_30": 43.0, "spr_thai_elderly_th_30": 39.2708, "spr_thai_ser_30": 39.2052, "spr_voxvietnam_vi_30": 37.5, "average": 41.1912, "spr_sfdusc_30": 43.7 }, { "model": "MERaLiON-3-3B-ASR", "modelLink": "https://huggingface.co/MERaLiON/MERaLiON-3-3B-ASR", "spr_emota_ta_30": 33.27991453, "spr_mig_my_30": 46.95, "spr_thai_ser_30": 27.71213749, "spr_voxvietnam_vi_30": 8.55, "spr_sfdusc_30": 19.7, "average": 27.2384 }, { "model": "Gemma-SEA-LION-v4.5-E2B-IT", "spr_emota_ta_30": 17.78846154, "spr_esd_en_30": 21.4, "spr_smaldusc_30": 31.85, "spr_thai_elderly_th_30": 15.83333333, "spr_thai_ser_30": 21.32116004, "spr_voxvietnam_vi_30": 24.0, "average": 21.7992, "spr_esd_zh_30": 24.95, "spr_mig_my_30": 16.5, "modelLink": "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4.5-E2B-IT", "spr_sfdusc_30": 22.55 } ], "metric": "accuracy", "metricInfo": "Accuracy - The higher, the better.", "ascending": false } } ], "metricsInfo": { "wer": "Word Error Rate (WER) - The Lower, the better.", "llama3_70b_judge_binary": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "llama3_70b_judge": "Model-as-a-Judge Peformance. Using LLAMA-3-70B. Scale from 0-100. The higher, the better.", "meteor": "METEOR Score. The higher, the better.", "bleu": "BLEU Score. The higher, the better." } }