{"url":"/task/spoken-language-understanding","name":"Spoken Language Understanding","slug":"spoken-language-understanding","description_markdown":null,"categories":[{"name":"Speech","url":"/area/speech"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":550,"papers_with_code":135,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":14,"subtasks":2,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/spoken-language-understanding-on-fluent","slug":"spoken-language-understanding-on-fluent","dataset":"Fluent Speech Commands","dataset_url":"/dataset/fluent-speech-commands","rows_in_archive":17,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"Finstreder (Conformer + AMT, character-based)","paper_title":"Finstreder: Simple and fast Spoken Language Understanding with Finite State Transducers using modern Speech-to-Text models","paper_url":"/paper/finstreder-simple-and-fast-spoken-language","paper_date":"2022-06-29","arxiv_id":"2206.14589","code_links":[{"title":"Jaco-Assistant/Jaco-Master","url":"https://gitlab.com/Jaco-Assistant/Jaco-Master"},{"title":"Jaco-Assistant/finstreder","url":"https://gitlab.com/Jaco-Assistant/finstreder"}],"syntology":null}},{"leaderboard":"/sota/spoken-language-understanding-on-snips","slug":"spoken-language-understanding-on-snips","dataset":"Snips-SmartLights","dataset_url":"/dataset/snips-smartlights","rows_in_archive":7,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"Finstreder (Conformer, character-based)","paper_title":"Finstreder: Simple and fast Spoken Language Understanding with Finite State Transducers using modern Speech-to-Text models","paper_url":"/paper/finstreder-simple-and-fast-spoken-language","paper_date":"2022-06-29","arxiv_id":"2206.14589","code_links":[{"title":"Jaco-Assistant/Jaco-Master","url":"https://gitlab.com/Jaco-Assistant/Jaco-Master"},{"title":"Jaco-Assistant/finstreder","url":"https://gitlab.com/Jaco-Assistant/finstreder"}],"syntology":null}},{"leaderboard":"/sota/spoken-language-understanding-on-snips-1","slug":"spoken-language-understanding-on-snips-1","dataset":"Snips-SmartSpeaker","dataset_url":"/dataset/snips-smartspeaker","rows_in_archive":5,"metrics":["Accuracy-EN (%)","Accuracy-FR (%)"],"first_row_in_archive_order":{"model":"Finstreder (Conformer, character-based)","paper_title":"Finstreder: Simple and fast Spoken Language Understanding with Finite State Transducers using modern Speech-to-Text models","paper_url":"/paper/finstreder-simple-and-fast-spoken-language","paper_date":"2022-06-29","arxiv_id":"2206.14589","code_links":[{"title":"Jaco-Assistant/Jaco-Master","url":"https://gitlab.com/Jaco-Assistant/Jaco-Master"},{"title":"Jaco-Assistant/finstreder","url":"https://gitlab.com/Jaco-Assistant/finstreder"}],"syntology":null}},{"leaderboard":"/sota/spoken-language-understanding-on-spoken-squad","slug":"spoken-language-understanding-on-spoken-squad","dataset":"Spoken-SQuAD","dataset_url":"/dataset/spoken-squad","rows_in_archive":4,"metrics":["F1 score"],"first_row_in_archive_order":{"model":"ALBERT","paper_title":"End-to-end Spoken Conversational Question Answering: Task, Dataset and Model","paper_url":"/paper/end-to-end-spoken-conversational-question","paper_date":"2022-04-29","arxiv_id":"2204.14272","code_links":[],"syntology":null}},{"leaderboard":"/sota/spoken-language-understanding-on-timers-and","slug":"spoken-language-understanding-on-timers-and","dataset":"Timers and Such","dataset_url":"/dataset/timers-and-such","rows_in_archive":3,"metrics":["Accuracy (%)"],"first_row_in_archive_order":{"model":"Finstreder (Conformer)","paper_title":"Finstreder: Simple and fast Spoken Language Understanding with Finite State Transducers using modern Speech-to-Text models","paper_url":"/paper/finstreder-simple-and-fast-spoken-language","paper_date":"2022-06-29","arxiv_id":"2206.14589","code_links":[{"title":"Jaco-Assistant/Jaco-Master","url":"https://gitlab.com/Jaco-Assistant/Jaco-Master"},{"title":"Jaco-Assistant/finstreder","url":"https://gitlab.com/Jaco-Assistant/finstreder"}],"syntology":null}}],"datasets":[{"url":"/dataset/slurp","name":"SLURP","full_name":"Spoken Language Understanding Resource Package","num_papers_in_archive":106},{"url":"/dataset/fluent-speech-commands","name":"Fluent Speech Commands","full_name":"","num_papers_in_archive":57},{"url":"/dataset/dialogue-state-tracking-challenge","name":"Dialogue State Tracking Challenge","full_name":"Dialogue State Tracking Challenge","num_papers_in_archive":33},{"url":"/dataset/spoken-squad","name":"Spoken-SQuAD","full_name":null,"num_papers_in_archive":24},{"url":"/dataset/xsid","name":"xSID","full_name":"Cross-lingual Slot and Intent Detection","num_papers_in_archive":18},{"url":"/dataset/snips-smartlights","name":"Snips-SmartLights","full_name":"","num_papers_in_archive":8},{"url":"/dataset/media","name":"MEDIA","full_name":"MEDIA","num_papers_in_archive":6},{"url":"/dataset/timers-and-such","name":"Timers and Such","full_name":"","num_papers_in_archive":5},{"url":"/dataset/snips-smartspeaker","name":"Snips-SmartSpeaker","full_name":"","num_papers_in_archive":3},{"url":"/dataset/almawave-slu","name":"Almawave-SLU","full_name":"","num_papers_in_archive":2},{"url":"/dataset/cqr","name":"CQR","full_name":"Contextual Query Rewrite","num_papers_in_archive":2},{"url":"/dataset/proslu","name":"ProSLU","full_name":"Profile-based Spoken Language Understanding","num_papers_in_archive":2},{"url":"/dataset/skit-s2i","name":"Skit-S2I","full_name":"Skit-S2I: An Indian Accented Speech to Intent dataset","num_papers_in_archive":1},{"url":"/dataset/vlogqa","name":"VlogQA","full_name":"Vietnamese Spoken-Based Machine Reading Comprehension","num_papers_in_archive":1}],"subtasks":[{"url":"/task/speech-tokenization","name":"Speech Tokenization"},{"url":"/task/spoken-language-identification","name":"Spoken language identification"}],"parent_tasks":[{"url":"/task/dialogue-understanding","name":"Dialogue Understanding"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":135,"tagged_in_all":550,"items":[{"url":"/paper/snips-voice-platform-an-embedded-spoken","title":"Snips Voice Platform: an embedded Spoken Language Understanding system for private-by-design voice interfaces","date":"2018-05-25","arxiv_id":"1805.10190","repositories_listed":16,"syntology":{"n":15,"n_ran":1,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/sdnet-contextualized-attention-based-deep","title":"SDNet: Contextualized Attention-based Deep Network for Conversational Question Answering","date":"2018-12-10","arxiv_id":"1812.03593","repositories_listed":6,"syntology":null},{"url":"/paper/speechbrain-a-general-purpose-speech-toolkit","title":"SpeechBrain: A General-Purpose Speech Toolkit","date":"2021-06-08","arxiv_id":"2106.04624","repositories_listed":4,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/branchformer-parallel-mlp-attention","title":"Branchformer: Parallel MLP-Attention Architectures to Capture Local and Global Context for Speech Recognition and Understanding","date":"2022-07-06","arxiv_id":"2207.02971","repositories_listed":3,"syntology":null},{"url":"/paper/using-speech-synthesis-to-train-end-to-end","title":"Using Speech Synthesis to Train End-to-End Spoken Language Understanding Models","date":"2019-10-21","arxiv_id":"1910.09463","repositories_listed":3,"syntology":{"n":12,"n_ran":1,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/spoken-squad-a-study-of-mitigating-the-impact","title":"Spoken SQuAD: A Study of Mitigating the Impact of Speech Recognition Errors on Listening Comprehension","date":"2018-04-01","arxiv_id":"1804.00320","repositories_listed":3,"syntology":null},{"url":"/paper/mmsu-a-massive-multi-task-spoken-language","title":"MMSU: A Massive Multi-task Spoken Language Understanding and Reasoning Benchmark","date":"2025-06-05","arxiv_id":"2506.04779","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_unverified":2,"n_pointer_only":6}},{"url":"/paper/retqa-a-large-scale-open-domain-tabular","title":"RETQA: A Large-Scale Open-Domain Tabular Question Answering Dataset for Real Estate Sector","date":"2024-12-13","arxiv_id":"2412.10104","repositories_listed":2,"syntology":null},{"url":"/paper/back-transcription-as-a-method-for-evaluating","title":"Back Transcription as a Method for Evaluating Robustness of Natural Language Understanding Models to Speech Recognition Errors","date":"2023-10-25","arxiv_id":"2310.16609","repositories_listed":2,"syntology":null},{"url":"/paper/lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/a-comparative-study-on-e-branchformer-vs","title":"A Comparative Study on E-Branchformer vs Conformer in Speech Recognition, Translation, and Understanding Tasks","date":"2023-05-18","arxiv_id":"2305.11073","repositories_listed":2,"syntology":null},{"url":"/paper/finstreder-simple-and-fast-spoken-language","title":"Finstreder: Simple and fast Spoken Language Understanding with Finite State Transducers using modern Speech-to-Text models","date":"2022-06-29","arxiv_id":"2206.14589","repositories_listed":2,"syntology":null},{"url":"/paper/espnet-slu-advancing-spoken-language","title":"ESPnet-SLU: Advancing Spoken Language Understanding through ESPnet","date":"2021-11-29","arxiv_id":"2111.14706","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/from-masked-language-modeling-to-translation","title":"From Masked Language Modeling to Translation: Non-English Auxiliary Tasks Improve Zero-shot Spoken Language Understanding","date":"2021-05-15","arxiv_id":"2105.07316","repositories_listed":2,"syntology":null},{"url":"/paper/timers-and-such-a-practical-benchmark-for","title":"Timers and Such: A Practical Benchmark for Spoken Language Understanding with Numbers","date":"2021-04-04","arxiv_id":"2104.01604","repositories_listed":2,"syntology":null},{"url":"/paper/semi-supervised-speech-language-joint-pre-1","title":"SPLAT: Speech-Language Joint Pre-Training for Spoken Language Understanding","date":"2020-10-05","arxiv_id":"2010.02295","repositories_listed":2,"syntology":null},{"url":"/paper/learning-spoken-language-representations-with-1","title":"Learning Spoken Language Representations with Neural Lattice Language Modeling","date":"2020-07-06","arxiv_id":"2007.02629","repositories_listed":2,"syntology":null},{"url":"/paper/enriching-existing-conversational-emotion","title":"EDA: Enriching Emotional Dialogue Acts using an Ensemble of Neural Annotators","date":"2019-12-02","arxiv_id":"1912.00819","repositories_listed":2,"syntology":null},{"url":"/paper/cm-net-a-novel-collaborative-memory-network","title":"CM-Net: A Novel Collaborative Memory Network for Spoken Language Understanding","date":"2019-09-16","arxiv_id":"1909.06937","repositories_listed":2,"syntology":null},{"url":"/paper/a-stack-propagation-framework-with-token","title":"A Stack-Propagation Framework with Token-Level Intent Detection for Spoken Language Understanding","date":"2019-09-05","arxiv_id":"1909.02188","repositories_listed":2,"syntology":null},{"url":"/paper/a-novel-bi-directional-interrelated-model-for","title":"A Novel Bi-directional Interrelated Model for Joint Intent Detection and Slot Filling","date":"2019-06-30","arxiv_id":"1907.00390","repositories_listed":2,"syntology":null},{"url":"/paper/mitigating-the-impact-of-speech-recognition-1","title":"Mitigating the Impact of Speech Recognition Errors on Spoken Question Answering by Adversarial Domain Adaptation","date":"2019-04-16","arxiv_id":"1904.07904","repositories_listed":2,"syntology":null},{"url":"/paper/speech-model-pre-training-for-end-to-end","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","date":"2019-04-07","arxiv_id":"1904.03670","repositories_listed":2,"syntology":null},{"url":"/paper/spoken-language-understanding-on-the-edge","title":"Spoken Language Understanding on the Edge","date":"2018-10-30","arxiv_id":"1810.12735","repositories_listed":2,"syntology":{"n":14,"n_ran":1,"n_unverified":13,"n_pointer_only":0}},{"url":"/paper/slot-gated-modeling-for-joint-slot-filling","title":"Slot-Gated Modeling for Joint Slot Filling and Intent Prediction","date":"2018-06-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/kan-you-hear-me-exploring-kolmogorov-arnold","title":"\"KAN you hear me?\" Exploring Kolmogorov-Arnold Networks for Spoken Language Understanding","date":"2025-05-26","arxiv_id":"2505.20176","repositories_listed":1,"syntology":null},{"url":"/paper/alexa-can-you-forget-me-machine-unlearning","title":"\"Alexa, can you forget me?\" Machine Unlearning Benchmark in Spoken Language Understanding","date":"2025-05-21","arxiv_id":"2505.15700","repositories_listed":1,"syntology":null},{"url":"/paper/quads-quantized-distillation-framework-for","title":"QUADS: QUAntized Distillation Framework for Efficient Speech Language Understanding","date":"2025-05-19","arxiv_id":"2505.14723","repositories_listed":1,"syntology":null},{"url":"/paper/livelongbench-tackling-long-context","title":"LiveLongBench: Tackling Long-Context Understanding for Spoken Texts from Live Streams","date":"2025-04-24","arxiv_id":"2504.17366","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-the-effect-of-transcription-noise","title":"Measuring the Effect of Transcription Noise on Downstream Language Understanding Tasks","date":"2025-02-19","arxiv_id":"2502.13645","repositories_listed":1,"syntology":null}],"syntology_records":7,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}