{"url":"/dataset/gigast","name":"GigaST","full_name":null,"description_markdown":"GigaST is a large-scale pseudo speech translation (ST) corpus. The corpus was created by translating the text in GigaSpeech, an English ASR corpus, into German and Chinese. The training set is translated by a strong machine translation system and the test set was translated by human. ST models trained with an addition of the corpus obtain new state-of-the-art results on the MuST-C English-German benchmark test set.","description_withheld":null,"homepage":"https://st-benchmark.github.io/resources/GigaST.html","introduced_date":"2022-04-08","introduced_date_note":null,"introduced_by":{"paper":"/paper/gigast-a-10000-hour-pseudo-speech-translation","title":"GigaST: A 10,000-hour Pseudo Speech Translation Corpus","first_author":"Rong Ye","url":null},"license":{"name":"CC BY-NC 4.0","url":"https://creativecommons.org/licenses/by-nc/4.0/legalcode"},"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Machine Translation","url":"/task/machine-translation","datasets_with_task":"/datasets/task/machine-translation"},{"name":"Translation","url":"/task/translation","datasets_with_task":"/datasets/task/translation"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["GigaST"],"data_loaders":[],"num_papers_in_archive":8,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-25T09:33:49+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}