{"url":"/dataset/crowdspeech","name":"CrowdSpeech","full_name":null,"description_markdown":"**CrowdSpeech** is a publicly available large-scale dataset of crowdsourced audio transcriptions. It contains annotations for more than 20 hours of English speech from more than 1,000 crowd workers.","description_withheld":null,"homepage":"https://github.com/pilot7747/VoxDIY","introduced_date":"2021-07-02","introduced_date_note":null,"introduced_by":{"paper":"/paper/vox-populi-vox-diy-benchmark-dataset-for","title":"CrowdSpeech and VoxDIY: Benchmark Datasets for Crowdsourced Audio Transcription","first_author":"Nikita Pavlichenko","url":null},"license":{"name":"Attribution 4.0 International","url":"https://github.com/pilot7747/VoxDIY/blob/main/data/LICENSE"},"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Crowdsourced Text Aggregation","url":"/task/crowdsourced-text-aggregation","datasets_with_task":"/datasets/task/crowdsourced-text-aggregation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["CrowdSpeech","CrowdSpeech test-clean","CrowdSpeech test-other"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/toloka/CrowdSpeech","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":1,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/crowdsourced-text-aggregation-on-crowdspeech","task":"Crowdsourced Text Aggregation","dataset_variant":"CrowdSpeech test-clean","rows":3,"metrics":["Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"ROVER","paper":"/paper/vox-populi-vox-diy-benchmark-dataset-for","metrics":{"Word Error Rate (WER)":"7.29"},"code_links":[{"title":"Toloka/CrowdSpeech","url":"https://github.com/Toloka/CrowdSpeech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/crowdsourced-text-aggregation-on-crowdspeech-1","task":"Crowdsourced Text Aggregation","dataset_variant":"CrowdSpeech test-other","rows":3,"metrics":["Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"ROVER","paper":"/paper/vox-populi-vox-diy-benchmark-dataset-for","metrics":{"Word Error Rate (WER)":"13.41"},"code_links":[{"title":"Toloka/CrowdSpeech","url":"https://github.com/Toloka/CrowdSpeech"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/vox-populi-vox-diy-benchmark-dataset-for","title":"CrowdSpeech and VoxDIY: Benchmark Datasets for Crowdsourced Audio Transcription","date":"2021-07-02","rows_on_this_dataset":6,"code_links":1,"syntology":null}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}