{"url":"/dataset/gigaspeech","name":"GigaSpeech","full_name":null,"description_markdown":"GigaSpeech, an evolving, multi-domain English speech recognition corpus with 10,000 hours of high quality labeled audio suitable for supervised training, and 40,000 hours of total audio suitable for semi-supervised and unsupervised training.","description_withheld":null,"homepage":"https://github.com/SpeechColab/GigaSpeech","introduced_date":"2021-06-13","introduced_date_note":null,"introduced_by":{"paper":"/paper/gigaspeech-an-evolving-multi-domain-asr","title":"GigaSpeech: An Evolving, Multi-domain ASR Corpus with 10,000 Hours of Transcribed Audio","first_author":"Guoguo Chen","url":null},"license":{"name":"Unknown","url":null},"modalities":[{"name":"Audio","url":"/datasets/modality/audio"},{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Automatic Speech Recognition","url":"/task/automatic-speech-recognition-2","datasets_with_task":"/datasets/task/automatic-speech-recognition-2"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["GigaSpeech","GigaSpeech DEV","GigaSpeech TEST"],"data_loaders":[{"repo":"https://github.com/kadirnar/whisper-plus","url":"https://github.com/kadirnar/whisper-plus","frameworks":["pytorch"]}],"num_papers_in_archive":87,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-recognition-on-gigaspeech-dev","task":"Speech Recognition","dataset_variant":"GigaSpeech DEV","rows":5,"metrics":["Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"SAMBA ASR","paper":"/paper/samba-asr-state-of-the-art-speech-recognition","metrics":{"Word Error Rate (WER)":"9.12"},"code_links":[]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-recognition-on-gigaspeech-test","task":"Speech Recognition","dataset_variant":"GigaSpeech TEST","rows":5,"metrics":["Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"Zipformer+pruned transducer w/ CR-CTC\n(no external language model)","paper":"/paper/cr-ctc-consistency-regularization-on-ctc-for","metrics":{"Word Error Rate (WER)":"10.03"},"code_links":[{"title":"k2-fsa/icefall","url":"https://github.com/k2-fsa/icefall"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-recognition-on-gigaspeech","task":"Speech Recognition","dataset_variant":"GigaSpeech","rows":1,"metrics":["Word Error Rate (WER)"],"first_row_in_archive_order":{"model":"Conformer/Transformer-AED","paper":"/paper/gigaspeech-an-evolving-multi-domain-asr","metrics":{"Word Error Rate (WER)":"10.90"},"code_links":[{"title":"SpeechColab/GigaSpeech","url":"https://github.com/SpeechColab/GigaSpeech"},{"title":"speechtranslation/gigas2s","url":"https://github.com/speechtranslation/gigas2s"},{"title":"maikezuefle/contr-pretraining","url":"https://github.com/maikezuefle/contr-pretraining"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/samba-asr-state-of-the-art-speech-recognition","title":"Samba-ASR: State-Of-The-Art Speech Recognition Leveraging Structured State-Space Models","date":"2025-01-06","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cr-ctc-consistency-regularization-on-ctc-for","title":"CR-CTC: Consistency regularization on CTC for improved speech recognition","date":"2024-10-07","rows_on_this_dataset":7,"code_links":1,"syntology":null},{"paper":"/paper/gigaspeech-an-evolving-multi-domain-asr","title":"GigaSpeech: An Evolving, Multi-domain ASR Corpus with 10,000 Hours of Transcribed Audio","date":"2021-06-13","rows_on_this_dataset":3,"code_links":3,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":10,"samples_ran":2,"samples_unverified":8,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}