{"url":"/dataset/fleurs","name":"FLEURS","full_name":"Few-shot Learning Evaluation of Universal Representations of Speech","description_markdown":"We introduce FLEURS, the Few-shot Learning Evaluation of Universal Representations of Speech benchmark. FLEURS is an n-way parallel speech dataset in 102 languages built on top of the machine translation FLoRes-101 benchmark, with approximately 12 hours of speech supervision per language. FLEURS can be used for a variety of speech tasks, including Automatic Speech Recognition (ASR), Speech Language Identification (Speech LangID), Translation and Retrieval. In this paper, we provide baselines for the tasks based on multilingual pre-trained models like mSLAM. The goal of FLEURS is to enable speech technology in more languages and catalyze research in low-resource speech understanding.","description_withheld":null,"homepage":"https://huggingface.co/datasets/google/fleurs","introduced_date":"2022-05-25","introduced_date_note":null,"introduced_by":{"paper":"/paper/fleurs-few-shot-learning-evaluation-of","title":"FLEURS: Few-shot Learning Evaluation of Universal Representations of Speech","first_author":"Alexis Conneau","url":null},"license":{"name":"CC-BY","url":"https://creativecommons.org/licenses/by/2.5/"},"modalities":[{"name":"Texts","url":"/datasets/modality/texts"},{"name":"Audio","url":"/datasets/modality/audio"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Few-Shot Learning","url":"/task/few-shot-learning","datasets_with_task":"/datasets/task/few-shot-learning"},{"name":"Automatic Speech Recognition","url":"/task/automatic-speech-recognition-2","datasets_with_task":"/datasets/task/automatic-speech-recognition-2"},{"name":"Automatic Speech Recognition (ASR)","url":"/task/automatic-speech-recognition","datasets_with_task":"/datasets/task/automatic-speech-recognition"},{"name":"Spoken language identification","url":"/task/spoken-language-identification","datasets_with_task":"/datasets/task/spoken-language-identification"},{"name":"automatic-speech-translation","url":"/task/automatic-speech-translation","datasets_with_task":"/datasets/task/automatic-speech-translation"},{"name":"Natural Language Inference (Few-Shot)","url":"/task/natural-language-inference-few-shot","datasets_with_task":"/datasets/task/natural-language-inference-few-shot"}],"languages":[],"variants":["FLUERS Korean","google/fleurs tr_tr","google/fleurs tg_tj","google/fleurs ta_in","google/fleurs ro","google/fleurs pt_br","GOOGLE/FLEURS - PS_AF","google/fleurs ps_af","google/fleurs ko_kr","google/fleurs ja_jp","google/fleurs id_id","google/fleurs he_il","google/fleurs gl_es","google/fleurs cmn_hans_cn","google/fleurs ca","google/fleurs am_et","Google FLEURS","google/fleurs","FLEURS ASR","ERR2020, Common Voice 11.0, FLEURS","Fleurs (English)","FLEURS"],"data_loaders":[],"num_papers_in_archive":141,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/speech-recognition-on-fleurs","task":"Speech Recognition","dataset_variant":"FLEURS","rows":0,"metrics":["Test WER"],"first_row_in_archive_order":null,"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}