{"url":"/dataset/vocalset","name":"VocalSet","full_name":"VocalSet: A Singing Voice Dataset","description_markdown":"VocalSet is a a singing voice dataset consisting of 10.1 hours of monophonic recorded audio of professional singers demonstrating both standard and extended vocal techniques on all 5 vowels. Existing singing voice datasets aim to capture a focused subset of singing voice characteristics, and generally consist of just a few singers. VocalSet contains recordings from 20 different singers (9 male, 11 female) and a range of voice types.  VocalSet aims to improve the state of existing singing voice datasets and singing voice research by capturing not only a range of vowels, but also a diverse set of voices on many different vocal techniques, sung in contexts of scales, arpeggios, long tones, and excerpts.","description_withheld":null,"homepage":"","introduced_date":"2018-09-25","introduced_date_note":null,"introduced_by":{"paper":"/paper/vocalset-a-singing-voice-dataset","title":"VocalSet: A Singing Voice Dataset","first_author":"Julia Wilkins","url":null},"license":{"name":"Creative Commons Attribution 4.0 International","url":"https://zenodo.org/record/1442513"},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Singer Identification","url":"/task/singer-identification","datasets_with_task":"/datasets/task/singer-identification"},{"name":"Vocal technique classification","url":"/task/vocal-technique-classification","datasets_with_task":"/datasets/task/vocal-technique-classification"}],"languages":[],"variants":["VocalSet"],"data_loaders":[{"repo":"https://github.com/PeerapolUtt/Singer","url":"https://github.com/PeerapolUtt/Singer","frameworks":[]}],"num_papers_in_archive":30,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/singer-identification-on-vocalset-1","task":"Singer Identification","dataset_variant":"VocalSet","rows":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"M2D2 AS+","paper":"/paper/m2d2-exploring-general-purpose-audio-language","metrics":{"Accuracy":"92.7"},"code_links":[{"title":"nttcslab/m2d","url":"https://github.com/nttcslab/m2d"},{"title":"nttcslab/eval-audio-repr","url":"https://github.com/nttcslab/eval-audio-repr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/vocal-technique-classification-on-vocalset-1","task":"Vocal technique classification","dataset_variant":"VocalSet","rows":2,"metrics":["Accuracy "],"first_row_in_archive_order":{"model":"M2D2 AS+","paper":"/paper/m2d2-exploring-general-purpose-audio-language","metrics":{"Accuracy ":"78.9"},"code_links":[{"title":"nttcslab/m2d","url":"https://github.com/nttcslab/m2d"},{"title":"nttcslab/eval-audio-repr","url":"https://github.com/nttcslab/eval-audio-repr"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/m2d2-exploring-general-purpose-audio-language","title":"M2D2: Exploring General-purpose Audio-Language Representations Beyond CLAP","date":"2025-03-28","rows_on_this_dataset":4,"code_links":2,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}