CoolFace
30 shown

datasets

Training and evaluation data, with the modality, task and licence stated up front. Listed live from the Hugging Face Hub.

Clear all
01AudioLLMs /Multitask-National-Speech-Corpus-v1-extendaudio10M<n<100M5 likes6.3k downloads1y agoHugging Face02garak-llm /audio_achilles_heelaudion<1K1 likes3.7k downloads1y agoHugging Face03alvanlii /audio-llm-trainaudio1M<n<10M2 likes796 downloads2y agoHugging Face04Elfsong /musicai-background-music-audio-llm-benchmark Does Background Music Matter to Speech in Pre-trained Language Models The completed September 2026 study covers 8 model families, 55 instrumental recordings, and 10 evaluation settings. It studies how adding background music to the same spoken question changes model responses. Latest release and artifact guide Technical report PDF Complete LaTeX project LaTeX GitHub repository Matrices, figures, and supporting data Regenerated speech and mixtures: 550 archives / 250,800… See the full description on the dataset page: https://huggingface.co/datasets/Elfsong/musicai-background-music-audio-llm-benchmark.audio0 likes505 downloads6d agoHugging Face05AudioLLMs /aishell_1_zh_test@inproceedings{bu2017aishell, title={Aishell-1: An open-source mandarin speech corpus and a speech recognition baseline}, author={Bu, Hui and Du, Jiayu and Na, Xingyu and Wu, Bengu and Zheng, Hao}, booktitle={2017 20th conference of the oriental chapter of the international coordinating committee on speech databases and speech I/O systems and assessment (O-COCOSDA)}, pages={1--5}, year={2017}, organization={IEEE} } @article{wang2024audiobench, title={AudioBench: A Universal… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/aishell_1_zh_test.audio1K<n<10K1 likes482 downloads2y agoHugging Face06AudioLLMs /earnings22_test@article{del2022earnings, title={Earnings-22: A practical benchmark for accents in the wild}, author={Del Rio, Miguel and Ha, Peter and McNamara, Quinten and Miller, Corey and Chandra, Shipra}, journal={arXiv preprint arXiv:2203.15591}, year={2022} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou, Xunlong and Lin, Geyu and Sun, Shuo and Liu, Zhuohan and Zhang, Wenyu and Liu, Zhengyuan and Aw, AiTi… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/earnings22_test.audion<1K0 likes441 downloads2y agoHugging Face07AudioLLMs /spoken_squad_testThis dataset is licensed under the terms of the CC-BY-SA-4.0 license. https://github.com/Chia-Hsuan-Lee/Spoken-SQuAD/blob/master/LICENSE.md Author: @michaellee886 @article{li2018spoken, title={Spoken SQuAD: A study of mitigating the impact of speech recognition errors on listening comprehension}, author={Li, Chia-Hsuan and Wu, Szu-Lin and Liu, Chi-Liang and Lee, Hung-yi}, journal={arXiv preprint arXiv:1804.00320}, year={2018} } @article{wang2024audiobench, title={AudioBench: A… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/spoken_squad_test.audio1K<n<10K1 likes343 downloads1y agoHugging Face08AudioLLMs /MMAU-mini-do-not-useWARNING: The original dataset is revised and pleased refer to new data source. Please refer to: MMAU-v05.15.25: https://github.com/Sakshi113/MMAU @misc{sakshi2024mmaumassivemultitaskaudio, title={MMAU: A Massive Multi-Task Audio Understanding and Reasoning Benchmark}, author={S Sakshi and Utkarsh Tyagi and Sonal Kumar and Ashish Seth and Ramaneswaran Selvakumar and Oriol Nieto and Ramani Duraiswami and Sreyan Ghosh and Dinesh Manocha}, year={2024}, eprint={2410.19168}… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/MMAU-mini-do-not-use.audio1K<n<10K2 likes226 downloads1y agoHugging Face09AudioLLMs /gigaspeech_test@article{chen2021gigaspeech, title={Gigaspeech: An evolving, multi-domain asr corpus with 10,000 hours of transcribed audio}, author={Chen, Guoguo and Chai, Shuzhou and Wang, Guanbo and Du, Jiayu and Zhang, Wei-Qiang and Weng, Chao and Su, Dan and Povey, Daniel and Trmal, Jan and Zhang, Junbo and others}, journal={arXiv preprint arXiv:2106.06909}, year={2021} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/gigaspeech_test.audio10K<n<100K1 likes215 downloads2y agoHugging Face10AudioLLMs /peoples_speech_test@article{galvez2021people, title={The people's speech: A large-scale diverse english speech recognition dataset for commercial usage}, author={Galvez, Daniel and Diamos, Greg and Ciro, Juan and Cer{\'o}n, Juan Felipe and Achorn, Keith and Gopi, Anjali and Kanter, David and Lam, Maximilian and Mazumder, Mark and Reddi, Vijay Janapa}, journal={arXiv preprint arXiv:2111.09344}, year={2021} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/peoples_speech_test.audio10K<n<100K0 likes206 downloads2y agoHugging Face11AudioLLMs /librispeech_test_clean@inproceedings{panayotov2015librispeech, title={Librispeech: an asr corpus based on public domain audio books}, author={Panayotov, Vassil and Chen, Guoguo and Povey, Daniel and Khudanpur, Sanjeev}, booktitle={2015 IEEE international conference on acoustics, speech and signal processing (ICASSP)}, pages={5206--5210}, year={2015}, organization={IEEE} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/librispeech_test_clean.audio1K<n<10K3 likes187 downloads2y agoHugging Face12AudioLLMs /tedlium3_test@inproceedings{hernandez2018ted, title={TED-LIUM 3: Twice as much data and corpus repartition for experiments on speaker adaptation}, author={Hernandez, Fran{\c{c}}ois and Nguyen, Vincent and Ghannay, Sahar and Tomashenko, Natalia and Esteve, Yannick}, booktitle={Speech and Computer: 20th International Conference, SPECOM 2018, Leipzig, Germany, September 18--22, 2018, Proceedings 20}, pages={198--208}, year={2018}, organization={Springer} } @article{wang2024audiobench… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/tedlium3_test.audio1K<n<10K0 likes178 downloads2y agoHugging Face13AudioLLMs /meld_emotion_test@article{poria2018meld, title={Meld: A multimodal multi-party dataset for emotion recognition in conversations}, author={Poria, Soujanya and Hazarika, Devamanyu and Majumder, Navonil and Naik, Gautam and Cambria, Erik and Mihalcea, Rada}, journal={arXiv preprint arXiv:1810.02508}, year={2018} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou, Xunlong and Lin, Geyu and Sun, Shuo and Liu, Zhuohan and… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/meld_emotion_test.audio1K<n<10K1 likes170 downloads2y agoHugging Face14AudioLLMs /cn_college_listen_mcq_test@article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou, Xunlong and Lin, Geyu and Sun, Shuo and Liu, Zhuohan and Zhang, Wenyu and Liu, Zhengyuan and Aw, AiTi and Chen, Nancy F}, journal={NAACL}, year={2025} } audio1K<n<10K1 likes144 downloads2y agoHugging Face15AudioLLMs /clotho_aqa_test@inproceedings{lipping2022clotho, title={Clotho-aqa: A crowdsourced dataset for audio question answering}, author={Lipping, Samuel and Sudarsanam, Parthasaarathy and Drossos, Konstantinos and Virtanen, Tuomas}, booktitle={2022 30th European Signal Processing Conference (EUSIPCO)}, pages={1140--1144}, year={2022}, organization={IEEE} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou, Xunlong and… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/clotho_aqa_test.audio1K<n<10K0 likes144 downloads2y agoHugging Face16AudioLLMs /librispeech_test_other@inproceedings{panayotov2015librispeech, title={Librispeech: an asr corpus based on public domain audio books}, author={Panayotov, Vassil and Chen, Guoguo and Povey, Daniel and Khudanpur, Sanjeev}, booktitle={2015 IEEE international conference on acoustics, speech and signal processing (ICASSP)}, pages={5206--5210}, year={2015}, organization={IEEE} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/librispeech_test_other.audio1K<n<10K0 likes135 downloads2y agoHugging Face17AudioLLMs /earnings21_test@article{del2021earnings, title={Earnings-21: A practical benchmark for ASR in the wild}, author={Del Rio, Miguel and Delworth, Natalie and Westerman, Ryan and Huang, Michelle and Bhandari, Nishchal and Palakapilly, Joseph and McNamara, Quinten and Dong, Joshua and Zelasko, Piotr and Jett{\'e}, Miguel}, journal={arXiv preprint arXiv:2104.11348}, year={2021} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/earnings21_test.audion<1K0 likes133 downloads2y agoHugging Face18AudioLLMs /dream_tts_mcq_test@article{sun2019dream, title={DREAM: A challenge data set and models for dialogue-based reading comprehension}, author={Sun, Kai and Yu, Dian and Chen, Jianshu and Yu, Dong and Choi, Yejin and Cardie, Claire}, journal={Transactions of the Association for Computational Linguistics}, volume={7}, pages={217--231}, year={2019}, publisher={MIT Press One Rogers Street, Cambridge, MA 02142-1209, USA journals-info~…} } @article{wang2024audiobench, title={AudioBench: A Universal… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/dream_tts_mcq_test.audio1K<n<10K0 likes132 downloads2y agoHugging Face19AudioLLMs /audiocaps_test@inproceedings{kim2019audiocaps, title={Audiocaps: Generating captions for audios in the wild}, author={Kim, Chris Dongjoo and Kim, Byeongchang and Lee, Hyunmin and Kim, Gunhee}, booktitle={Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)}, pages={119--132}, year={2019} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/audiocaps_test.audio1K<n<10K2 likes129 downloads2y agoHugging Face20AudioLLMs /covost2_en_id_test@inproceedings{wang2021covost, title={CoVoST 2 and massively multilingual speech translation.}, author={Wang, Changhan and Wu, Anne and Gu, Jiatao and Pino, Juan}, booktitle={Interspeech}, volume={2021}, pages={2247--2251}, year={2021} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou, Xunlong and Lin, Geyu and Sun, Shuo and Liu, Zhuohan and Zhang, Wenyu and Liu, Zhengyuan and Aw, AiTi and Chen… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/covost2_en_id_test.audio10K<n<100K1 likes129 downloads2y agoHugging Face21AudioLLMs /common_voice_15_en_test@article{ardila2019common, title={Common voice: A massively-multilingual speech corpus}, author={Ardila, Rosana and Branson, Megan and Davis, Kelly and Henretty, Michael and Kohler, Michael and Meyer, Josh and Morais, Reuben and Saunders, Lindsay and Tyers, Francis M and Weber, Gregor}, journal={arXiv preprint arXiv:1912.06670}, year={2019} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou, Xunlong… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/common_voice_15_en_test.audio10K<n<100K1 likes119 downloads2y agoHugging Face22AudioLLMs /public_sg_speech_qa_test@article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou, Xunlong and Lin, Geyu and Sun, Shuo and Liu, Zhuohan and Zhang, Wenyu and Liu, Zhengyuan and Aw, AiTi and Chen, Nancy F}, journal={NAACL}, year={2025} } audion<1K0 likes116 downloads2y agoHugging Face23AudioLLMs /slue_p2_sqa5_test@article{shon2022slue, title={SLUE phase-2: A benchmark suite of diverse spoken language understanding tasks}, author={Shon, Suwon and Arora, Siddhant and Lin, Chyi-Jiunn and Pasad, Ankita and Wu, Felix and Sharma, Roshan and Wu, Wei-Lun and Lee, Hung-yi and Livescu, Karen and Watanabe, Shinji}, journal={arXiv preprint arXiv:2212.10525}, year={2022} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/slue_p2_sqa5_test.audion<1K0 likes115 downloads2y agoHugging Face24AudioLLMs /voxceleb_gender_test@article{nagrani2020voxceleb, title={Voxceleb: Large-scale speaker verification in the wild}, author={Nagrani, Arsha and Chung, Joon Son and Xie, Weidi and Zisserman, Andrew}, journal={Computer Speech \& Language}, volume={60}, pages={101027}, year={2020}, publisher={Elsevier} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou, Xunlong and Lin, Geyu and Sun, Shuo and Liu, Zhuohan and Zhang… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/voxceleb_gender_test.audio1K<n<10K0 likes108 downloads2y agoHugging Face25AudioLLMs /iemocap_emotion_recognition@article{busso2008iemocap, title={IEMOCAP: Interactive emotional dyadic motion capture database}, author={Busso, Carlos and Bulut, Murtaza and Lee, Chi-Chun and Kazemzadeh, Abe and Mower, Emily and Kim, Samuel and Chang, Jeannette N and Lee, Sungbok and Narayanan, Shrikanth S}, journal={Language resources and evaluation}, volume={42}, pages={335--359}, year={2008}, publisher={Springer} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/iemocap_emotion_recognition.audio1K<n<10K1 likes107 downloads2y agoHugging Face26AudioLLMs /gigaspeech2-test@article{yang2024gigaspeech, title={GigaSpeech 2: An Evolving, Large-Scale and Multi-domain ASR Corpus for Low-Resource Languages with Automated Crawling, Transcription and Refinement}, author={Yang, Yifan and Song, Zheshu and Zhuo, Jianheng and Cui, Mingyu and Li, Jinpeng and Yang, Bo and Du, Yexing and Ma, Ziyang and Liu, Xunying and Wang, Ziyuan and others}, journal={arXiv preprint arXiv:2406.11546}, year={2024} } @article{wang2024audiobench, title={AudioBench: A Universal… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/gigaspeech2-test.audio10K<n<100K1 likes103 downloads2y agoHugging Face27AudioLLMs /openhermes_instruction_test@article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou, Xunlong and Lin, Geyu and Sun, Shuo and Liu, Zhuohan and Zhang, Wenyu and Liu, Zhengyuan and Aw, AiTi and Chen, Nancy F}, journal={NAACL}, year={2025} } audion<1K2 likes101 downloads2y agoHugging Face28AudioLLMs /seame_dev_sgeThis is the dev dataset. @inproceedings{lyu2010seame, title={SEAME: a Mandarin-English code-switching speech corpus in south-east asia.}, author={Lyu, Dau-Cheng and Tan, Tien Ping and Chng, Engsiong and Li, Haizhou}, booktitle={Interspeech}, volume={10}, pages={1986--1989}, year={2010} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou, Xunlong and Lin, Geyu and Sun, Shuo and Liu, Zhuohan and… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/seame_dev_sge.audio1K<n<10K0 likes98 downloads2y agoHugging Face29AudioLLMs /voxceleb_accent_test@article{nagrani2020voxceleb, title={Voxceleb: Large-scale speaker verification in the wild}, author={Nagrani, Arsha and Chung, Joon Son and Xie, Weidi and Zisserman, Andrew}, journal={Computer Speech \& Language}, volume={60}, pages={101027}, year={2020}, publisher={Elsevier} } @article{wang2024audiobench, title={AudioBench: A Universal Benchmark for Audio Large Language Models}, author={Wang, Bin and Zou, Xunlong and Lin, Geyu and Sun, Shuo and Liu, Zhuohan and Zhang… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/voxceleb_accent_test.audio1K<n<10K0 likes97 downloads2y agoHugging Face30AudioLLMs /tedlium3_long_form_test@inproceedings{hernandez2018ted, title={TED-LIUM 3: Twice as much data and corpus repartition for experiments on speaker adaptation}, author={Hernandez, Fran{\c{c}}ois and Nguyen, Vincent and Ghannay, Sahar and Tomashenko, Natalia and Esteve, Yannick}, booktitle={Speech and Computer: 20th International Conference, SPECOM 2018, Leipzig, Germany, September 18--22, 2018, Proceedings 20}, pages={198--208}, year={2018}, organization={Springer} } @article{wang2024audiobench… See the full description on the dataset page: https://huggingface.co/datasets/AudioLLMs/tedlium3_long_form_test.audion<1K0 likes96 downloads2y agoHugging Face

Listings come live from the Hugging Face Hub API. CoolFace does not host these files.