{"url":"/dataset/slue","name":"SLUE","full_name":"Spoken Language Understanding Evaluation","description_markdown":"**Spoken Language Understanding Evaluation** (**SLUE**) is a suite of benchmark tasks  for spoken language understanding evaluation. It consists of limited-size labeled training sets and corresponding evaluation sets. This resource would allow the research community to track progress, evaluate pre-trained representations for higher-level tasks, and study open questions such as the utility of pipeline versus end-to-end approaches. The first phase of the SLUE benchmark suite consists of named entity recognition (NER), sentiment analysis (SA), and ASR on the corresponding datasets.\r\n\r\nCorpus includes:\r\n\r\n- SLUE-VoxPopuli: consists of ASR and NER tasks - [CC0 license](https://creativecommons.org/share-your-work/public-domain/cc0/)\r\n- SLUE-VoxCeleb: consists of ASR and SA tasks - [CCBY 4.0 license](https://creativecommons.org/licenses/by/4.0/)","description_withheld":null,"homepage":"https://github.com/asappresearch/slue-toolkit","introduced_date":"2021-11-19","introduced_date_note":null,"introduced_by":{"paper":"/paper/slue-new-benchmark-tasks-for-spoken-language","title":"SLUE: New Benchmark Tasks for Spoken Language Understanding Evaluation on Natural Speech","first_author":"Suwon Shon","url":null},"license":{"name":"Custom","url":null},"modalities":[{"name":"Speech","url":"/datasets/modality/speech"}],"tasks":[{"name":"Speech Recognition","url":"/task/speech-recognition","datasets_with_task":"/datasets/task/speech-recognition"},{"name":"Named Entity Recognition (NER)","url":"/task/named-entity-recognition-ner","datasets_with_task":"/datasets/task/named-entity-recognition-ner"},{"name":"Sentiment Analysis","url":"/task/sentiment-analysis","datasets_with_task":"/datasets/task/sentiment-analysis"}],"languages":[],"variants":["SLUE"],"data_loaders":[{"repo":"https://github.com/huggingface/datasets","url":"https://huggingface.co/datasets/asapp/slue","frameworks":["tf","pytorch","jax"]}],"num_papers_in_archive":22,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[{"leaderboard":"/sota/named-entity-recognition-on-slue","task":"Named Entity Recognition (NER)","dataset_variant":"SLUE","rows":13,"metrics":["F1 (%)","label-F1 (%)","Text model"],"first_row_in_archive_order":{"model":"W2V2-L-LL60K (pipeline approach, uses LM)","paper":"/paper/slue-new-benchmark-tasks-for-spoken-language","metrics":{"F1 (%)":"69.6","Text model":"DeBERTa-L","label-F1 (%)":"82.2"},"code_links":[{"title":"asappresearch/slue-toolkit","url":"https://github.com/asappresearch/slue-toolkit"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/sentiment-analysis-on-slue","task":"Sentiment Analysis","dataset_variant":"SLUE","rows":8,"metrics":["Recall (%)\t","F1 (%)","Text model"],"first_row_in_archive_order":{"model":"W2V2-L-LL60K (pipeline approach, uses LM)","paper":"/paper/slue-new-benchmark-tasks-for-spoken-language","metrics":{"F1 (%)":"63.3","Recall (%)\t":"60.4","Text model":"DeBERTa-L"},"code_links":[{"title":"asappresearch/slue-toolkit","url":"https://github.com/asappresearch/slue-toolkit"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"},{"leaderboard":"/sota/speech-recognition-on-slue","task":"Speech Recognition","dataset_variant":"SLUE","rows":8,"metrics":["VoxPopuli (Dev)","VoxPopuli (Test)","VoxCeleb (Dev)","VoxCeleb (Test)"],"first_row_in_archive_order":{"model":"W2V2-L-LL60K (+ TED-LIUM 3 LM)","paper":"/paper/slue-new-benchmark-tasks-for-spoken-language","metrics":{"VoxCeleb (Dev)":"9.1","VoxCeleb (Test)":"10.8","VoxPopuli (Dev)":"9.1","VoxPopuli (Test)":"9.3"},"code_links":[{"title":"asappresearch/slue-toolkit","url":"https://github.com/asappresearch/slue-toolkit"}]},"note":"rows are the archive's own order at snapshot; nothing here re-ranks them"}],"papers_with_a_benchmark_row":[{"paper":"/paper/wav2seq-pre-training-speech-to-text-encoder","title":"Wav2Seq: Pre-training Speech-to-Text Encoder-Decoder Models Using Pseudo Languages","date":"2022-05-02","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/slue-new-benchmark-tasks-for-spoken-language","title":"SLUE: New Benchmark Tasks for Spoken Language Understanding Evaluation on Natural Speech","date":"2021-11-19","rows_on_this_dataset":28,"code_links":1,"syntology":{"read_at":"2026-09-24T18:15:14+00:00","samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim."}}],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":1,"samples_harvested":12,"samples_ran":0,"samples_unverified":12,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":1,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}