{"url":"/dataset/vocalsound","name":"VocalSound","full_name":null,"description_markdown":"VocalSound is a free dataset consisting of 21,024 crowdsourced recordings of laughter, sighs, coughs, throat clearing, sneezes, and sniffs from 3,365 unique subjects. The VocalSound dataset also contains meta-information such as speaker age, gender, native language, country, and health condition.","description_withheld":null,"homepage":"https://groups.csail.mit.edu/sls/downloads/vocalsound/","introduced_date":"2022-05-06","introduced_date_note":null,"introduced_by":{"paper":"/paper/vocalsound-a-dataset-for-improving-human","title":"Vocalsound: A Dataset for Improving Human Vocal Sounds Recognition","first_author":"Yuan Gong","url":null},"license":{"name":"CC BY-SA","url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode"},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Audio Classification","url":"/task/audio-classification","datasets_with_task":"/datasets/task/audio-classification"},{"name":"Audio Tagging","url":"/task/audio-tagging","datasets_with_task":"/datasets/task/audio-tagging"}],"languages":[],"variants":["VocalSound"],"data_loaders":[{"repo":"https://github.com/YuanGongND/vocalsound","url":"https://github.com/YuanGongND/vocalsound","frameworks":["pytorch"]}],"num_papers_in_archive":24,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/audio-classification-on-vocalsound","task":"Audio Classification","dataset_variant":"VocalSound","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"VocalSound Baseline","paper":"/paper/vocalsound-a-dataset-for-improving-human","metrics":{"Accuracy ":"90.5"},"code_links":[{"title":"YuanGongND/vocalsound","url":"https://github.com/YuanGongND/vocalsound"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/qwen-audio-advancing-universal-audio","title":"Qwen-Audio: Advancing Universal Audio Understanding via Unified Large-Scale Audio-Language Models","date":"2023-11-14","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":7,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}},{"paper":"/paper/vocalsound-a-dataset-for-improving-human","title":"Vocalsound: A Dataset for Improving Human Vocal Sounds Recognition","date":"2022-05-06","rows_on_this_dataset":1,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":7,"samples_ran":5,"samples_unverified":2,"pointer_only_for_licence":7,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}