{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/spoken-language-understanding/papers/2","list_of":"/task/spoken-language-understanding","task":"Spoken Language Understanding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":6,"rows_per_page":100,"rows":[101,200],"of":550,"counts":{"archive_papers_tagged":550,"with_a_code_link":135,"where_syntology_ran_a_sample":12,"not_listed_spam_title":0,"listed":550,"listed_where_code_ran":12,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":11,"every_run_a_failure_of_syntologys_instrument":1,"listed_with_a_run_with_no_instrument_failure":11,"listed_every_run_a_failure_of_syntologys_instrument":1,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/spoken-language-understanding","prev":"/task/spoken-language-understanding","next":"/task/spoken-language-understanding/papers/3","papers":[{"url":"/paper/slurp-a-spoken-language-understanding-1","slug":"slurp-a-spoken-language-understanding-1","title":"SLURP: A Spoken Language Understanding Resource Package","date":"2020-11-26","arxiv_id":"2011.13205","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-pretrained-transformer-to-lattices","slug":"adapting-pretrained-transformer-to-lattices","title":"Adapting Pretrained Transformer to Lattices for Spoken Language Understanding","date":"2020-11-02","arxiv_id":"2011.00780","repositories_listed":1,"syntology":null},{"url":"/paper/semi-supervised-spoken-language-understanding","slug":"semi-supervised-spoken-language-understanding","title":"Semi-Supervised Spoken Language Understanding via Self-Supervised Speech and Language Model Pretraining","date":"2020-10-26","arxiv_id":"2010.13826","repositories_listed":1,"syntology":null},{"url":"/paper/two-stage-textual-knowledge-distillation-to","slug":"two-stage-textual-knowledge-distillation-to","title":"Two-stage Textual Knowledge Distillation for End-to-End Spoken Language Understanding","date":"2020-10-25","arxiv_id":"2010.13105","repositories_listed":1,"syntology":null},{"url":"/paper/a-co-interactive-transformer-for-joint-slot","slug":"a-co-interactive-transformer-for-joint-slot","title":"A Co-Interactive Transformer for Joint Slot Filling and Intent Detection","date":"2020-10-08","arxiv_id":"2010.03880","repositories_listed":1,"syntology":null},{"url":"/paper/injecting-word-information-with-multi-level","slug":"injecting-word-information-with-multi-level","title":"Injecting Word Information with Multi-Level Word Adapter for Chinese Spoken Language Understanding","date":"2020-10-08","arxiv_id":"2010.03903","repositories_listed":1,"syntology":null},{"url":"/paper/slotrefine-a-fast-non-autoregressive-model","slug":"slotrefine-a-fast-non-autoregressive-model","title":"SlotRefine: A Fast Non-Autoregressive Model for Joint Intent Detection and Slot Filling","date":"2020-10-06","arxiv_id":"2010.02693","repositories_listed":1,"syntology":null},{"url":"/paper/textual-supervision-for-visually-grounded","slug":"textual-supervision-for-visually-grounded","title":"Textual Supervision for Visually Grounded Spoken Language Understanding","date":"2020-10-06","arxiv_id":"2010.02806","repositories_listed":1,"syntology":null},{"url":"/paper/cross-lingual-spoken-language-understanding","slug":"cross-lingual-spoken-language-understanding","title":"Cross-lingual Spoken Language Understanding with Regularized Representation Alignment","date":"2020-09-30","arxiv_id":"2009.14510","repositories_listed":1,"syntology":null},{"url":"/paper/jointly-encoding-word-confusion-network-and","slug":"jointly-encoding-word-confusion-network-and","title":"Jointly Encoding Word Confusion Network and Dialogue Context with BERT for Spoken Language Understanding","date":"2020-05-24","arxiv_id":"2005.11640","repositories_listed":1,"syntology":null},{"url":"/paper/data-augmentation-for-spoken-language-1","slug":"data-augmentation-for-spoken-language-1","title":"Data Augmentation for Spoken Language Understanding via Pretrained Language Models","date":"2020-04-29","arxiv_id":"2004.13952","repositories_listed":1,"syntology":null},{"url":"/paper/td-gin-token-level-dynamic-graph-interactive","slug":"td-gin-token-level-dynamic-graph-interactive","title":"AGIF: An Adaptive Graph-Interactive Framework for Joint Multiple Intent Detection and Slot Filling","date":"2020-04-21","arxiv_id":"2004.10087","repositories_listed":1,"syntology":null},{"url":"/paper/fast-intent-classification-for-spoken","slug":"fast-intent-classification-for-spoken","title":"Fast Intent Classification for Spoken Language Understanding","date":"2019-12-03","arxiv_id":"1912.01728","repositories_listed":1,"syntology":null},{"url":"/paper/disambiguating-speech-intention-via-audio","slug":"disambiguating-speech-intention-via-audio","title":"Text Matters but Speech Influences: A Computational Analysis of Syntactic Ambiguity Resolution","date":"2019-10-21","arxiv_id":"1910.09275","repositories_listed":1,"syntology":null},{"url":"/paper/learning-asr-robust-contextualized-embeddings","slug":"learning-asr-robust-contextualized-embeddings","title":"Learning ASR-Robust Contextualized Embeddings for Spoken Language Understanding","date":"2019-09-24","arxiv_id":"1909.10861","repositories_listed":1,"syntology":null},{"url":"/paper/data-augmentation-with-atomic-templates-for","slug":"data-augmentation-with-atomic-templates-for","title":"Data Augmentation with Atomic Templates for Spoken Language Understanding","date":"2019-08-28","arxiv_id":"1908.10770","repositories_listed":1,"syntology":null},{"url":"/paper/energy-based-self-attentive-learning-of","slug":"energy-based-self-attentive-learning-of","title":"Energy-based Self-attentive Learning of Abstractive Communities for Spoken Language Understanding","date":"2019-04-20","arxiv_id":"1904.09491","repositories_listed":1,"syntology":null},{"url":"/paper/a-hierarchical-decoding-model-for-spoken","slug":"a-hierarchical-decoding-model-for-spoken","title":"A Hierarchical Decoding Model For Spoken Language Understanding From Unaligned Data","date":"2019-04-09","arxiv_id":"1904.04498","repositories_listed":1,"syntology":null},{"url":"/paper/spoken-language-intent-detection-using","slug":"spoken-language-intent-detection-using","title":"Spoken Language Intent Detection using Confusion2Vec","date":"2019-04-07","arxiv_id":"1904.03576","repositories_listed":1,"syntology":null},{"url":"/paper/a-dataset-for-resolving-referring-expressions","slug":"a-dataset-for-resolving-referring-expressions","title":"A dataset for resolving referring expressions in spoken dialogue via contextual query rewrites (CQR)","date":"2019-03-28","arxiv_id":"1903.11783","repositories_listed":1,"syntology":null},{"url":"/paper/decay-function-free-time-aware-attention-to","slug":"decay-function-free-time-aware-attention-to","title":"Decay-Function-Free Time-Aware Attention to Context and Speaker Indicator for Spoken Language Understanding","date":"2019-03-20","arxiv_id":"1903.08450","repositories_listed":1,"syntology":null},{"url":"/paper/audio-linguistic-embeddings-for-spoken","slug":"audio-linguistic-embeddings-for-spoken","title":"Audio-Linguistic Embeddings for Spoken Sentences","date":"2019-02-20","arxiv_id":"1902.07817","repositories_listed":1,"syntology":null},{"url":"/paper/a-bi-model-based-rnn-semantic-frame-parsing","slug":"a-bi-model-based-rnn-semantic-frame-parsing","title":"A Bi-model based RNN Semantic Frame Parsing Model for Intent Detection and Slot Filling","date":"2018-12-26","arxiv_id":"1812.10235","repositories_listed":1,"syntology":null},{"url":"/paper/recurrent-neural-networks-with-pre-trained","slug":"recurrent-neural-networks-with-pre-trained","title":"Recurrent Neural Networks with Pre-trained Language Model Embedding for Slot Filling Task","date":"2018-12-12","arxiv_id":"1812.05199","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-transfer-learning-for-spoken","slug":"unsupervised-transfer-learning-for-spoken","title":"Unsupervised Transfer Learning for Spoken Language Understanding in Intelligent Agents","date":"2018-11-13","arxiv_id":"1811.05370","repositories_listed":1,"syntology":null},{"url":"/paper/flowqa-grasping-flow-in-history-for","slug":"flowqa-grasping-flow-in-history-for","title":"FlowQA: Grasping Flow in History for Conversational Machine Comprehension","date":"2018-10-06","arxiv_id":"1810.06683","repositories_listed":1,"syntology":null},{"url":"/paper/fully-statistical-neural-belief-tracking-1","slug":"fully-statistical-neural-belief-tracking-1","title":"Fully Statistical Neural Belief Tracking","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/improving-slot-filling-in-spoken-language","slug":"improving-slot-filling-in-spoken-language","title":"Improving Slot Filling in Spoken Language Understanding with Joint Pointer and Attention","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/discourse-wizard-discovering-deep-discourse","slug":"discourse-wizard-discovering-deep-discourse","title":"Discourse-Wizard: Discovering Deep Discourse Structure in your Conversation with RNNs","date":"2018-06-29","arxiv_id":"1806.11420","repositories_listed":1,"syntology":null},{"url":"/paper/iso-standard-domain-independent-dialogue-act","slug":"iso-standard-domain-independent-dialogue-act","title":"ISO-Standard Domain-Independent Dialogue Act Tagging for Conversational Agents","date":"2018-06-12","arxiv_id":"1806.04327","repositories_listed":1,"syntology":null},{"url":"/paper/how-time-matters-learning-time-decay","slug":"how-time-matters-learning-time-decay","title":"How Time Matters: Learning Time-Decay Attention for Contextual Spoken Language Understanding in Dialogues","date":"2018-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/towards-end-to-end-spoken-language","slug":"towards-end-to-end-spoken-language","title":"Towards end-to-end spoken language understanding","date":"2018-02-23","arxiv_id":"1802.08395","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-time-aware-attention-to-speaker-roles","slug":"dynamic-time-aware-attention-to-speaker-roles","title":"Dynamic Time-Aware Attention to Speaker Roles and Contexts for Spoken Language Understanding","date":"2017-09-30","arxiv_id":"1710.00165","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-dialogue-context-modeling-for","slug":"sequential-dialogue-context-modeling-for","title":"Sequential Dialogue Context Modeling for Spoken Language Understanding","date":"2017-05-08","arxiv_id":"1705.03455","repositories_listed":1,"syntology":null},{"url":"/paper/graph-based-semi-supervised-conditional","slug":"graph-based-semi-supervised-conditional","title":"Graph-Based Semi-Supervised Conditional Random Fields For Spoken Language Understanding Using Unaligned Data","date":"2017-01-30","arxiv_id":"1701.08533","repositories_listed":1,"syntology":null},{"url":null,"slug":"alas-measuring-latent-speech-text-alignment","title":"ALAS: Measuring Latent Speech-Text Alignment For Spoken Language Understanding In Multimodal LLMs","date":"2025-05-26","arxiv_id":"2505.19937","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-the-effect-of-segmentation-and","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","date":"2025-05-23","arxiv_id":"2505.17446","repositories_listed":0,"syntology":null},{"url":null,"slug":"spoken-language-understanding-on-unseen-tasks","title":"Spoken Language Understanding on Unseen Tasks With In-Context Learning","date":"2025-05-12","arxiv_id":"2505.07731","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-transducer-based-spoken-language","title":"Improving Transducer-Based Spoken Language Understanding with Self-Conditioned CTC and Knowledge Transfer","date":"2025-01-03","arxiv_id":"2501.01936","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-and-discussion-of-the-suitability","title":"An Overview and Discussion of the Suitability of Existing Speech Datasets to Train Machine Learning Models for Collective Problem Solving","date":"2024-12-24","arxiv_id":"2412.18489","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-speech-large-language-models","title":"A Survey on Speech Large Language Models","date":"2024-10-24","arxiv_id":"2410.18908","repositories_listed":0,"syntology":null},{"url":null,"slug":"interventional-speech-noise-injection-for-asr","title":"Interventional Speech Noise Injection for ASR Generalizable Spoken Language Understanding","date":"2024-10-21","arxiv_id":"2410.15609","repositories_listed":0,"syntology":null},{"url":null,"slug":"speech-recognition-rescoring-with-large","title":"Speech Recognition Rescoring with Large Speech-Text Foundation Models","date":"2024-09-25","arxiv_id":"2409.16654","repositories_listed":0,"syntology":null},{"url":null,"slug":"increasing-faithfulness-in-human-human-dialog","title":"Increasing faithfulness in human-human dialog summarization with Spoken Language Understanding tasks","date":"2024-09-16","arxiv_id":"2409.10070","repositories_listed":0,"syntology":null},{"url":null,"slug":"clean-label-attacks-against-slu-systems","title":"Clean Label Attacks against SLU Systems","date":"2024-09-13","arxiv_id":"2409.08985","repositories_listed":0,"syntology":null},{"url":null,"slug":"whisma-a-speech-llm-to-perform-zero-shot","title":"WHISMA: A Speech-LLM to Perform Zero-shot Spoken Language Understanding","date":"2024-08-29","arxiv_id":"2408.16423","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-whisper-for-qa-driven-zero-shot-end","title":"Prompting Whisper for QA-driven Zero-shot End-to-end Spoken Language Understanding","date":"2024-06-21","arxiv_id":"2406.15209","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-task-specific-subnetworks-in-multi","title":"Finding Task-specific Subnetworks in Multi-task Spoken Language Understanding Model","date":"2024-06-18","arxiv_id":"2406.12317","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dual-task-learning-approach-to-fine-tune-a","title":"A dual task learning approach to fine-tune a multilingual semantic speech encoder for Spoken Language Understanding","date":"2024-06-17","arxiv_id":"2406.12141","repositories_listed":0,"syntology":null},{"url":null,"slug":"croprompt-cross-task-interactive-prompting","title":"CroPrompt: Cross-task Interactive Prompting for Zero-shot Spoken Language Understanding","date":"2024-06-15","arxiv_id":"2406.10505","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-evaluation-of-speech-foundation-models","title":"On the Evaluation of Speech Foundation Models for Spoken Language Understanding","date":"2024-06-14","arxiv_id":"2406.10083","repositories_listed":0,"syntology":null},{"url":null,"slug":"discreteslu-a-large-language-model-with-self","title":"DiscreteSLU: A Large Language Model with Self-Supervised Discrete Speech Units for Spoken Language Understanding","date":"2024-06-13","arxiv_id":"2406.09345","repositories_listed":0,"syntology":null},{"url":null,"slug":"prodeliberation-parallel-robust-deliberation","title":"PRoDeliberation: Parallel Robust Deliberation for End-to-End Spoken Language Understanding","date":"2024-06-12","arxiv_id":"2406.07823","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-spoken-language-understanding-via","title":"Towards Spoken Language Understanding via Multi-level Multi-grained Contrastive Learning","date":"2024-05-31","arxiv_id":"2405.20852","repositories_listed":0,"syntology":null},{"url":"/paper/msner-a-multilingual-speech-dataset-for-named","slug":"msner-a-multilingual-speech-dataset-for-named","title":"MSNER: A Multilingual Speech Dataset for Named Entity Recognition","date":"2024-05-19","arxiv_id":"2405.11519","repositories_listed":0,"syntology":null},{"url":null,"slug":"sonos-voice-control-bias-assessment-dataset-a","title":"Sonos Voice Control Bias Assessment Dataset: A Methodology for Demographic Bias Assessment in Voice Assistants","date":"2024-05-14","arxiv_id":"2405.19342","repositories_listed":0,"syntology":null},{"url":null,"slug":"hc-2-l-hybrid-and-cooperative-contrastive","title":"HC$^2$L: Hybrid and Cooperative Contrastive Learning for Cross-lingual Spoken Language Understanding","date":"2024-05-10","arxiv_id":"2405.06204","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-output-level-task-relatedness-in","title":"Modeling Output-Level Task Relatedness in Multi-Task Learning with Feedback Mechanism","date":"2024-04-01","arxiv_id":"2404.00885","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-end-to-end-spoken-language","title":"Privacy-Preserving End-to-End Spoken Language Understanding","date":"2024-03-22","arxiv_id":"2403.15510","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-has-lebenchmark-learnt-about-french","title":"What has LeBenchmark Learnt about French Syntax?","date":"2024-03-04","arxiv_id":"2403.02173","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-and-improving-continual-learning","title":"Evaluating and Improving Continual Learning in Spoken Language Understanding","date":"2024-02-16","arxiv_id":"2402.10427","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-balancing-act-unmasking-and-alleviating","title":"The Balancing Act: Unmasking and Alleviating ASR Biases in Portuguese","date":"2024-02-12","arxiv_id":"2402.07513","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-self-supervised-speech-model-with","title":"Integrating Self-supervised Speech Model with Pseudo Word-level Targets from Visually-grounded Speech Model","date":"2024-02-08","arxiv_id":"2402.05819","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-semantic-information-from-raw-audio","title":"Learning Semantic Information from Raw Audio Signal Using Both Contextual and Phonetic Representations","date":"2024-02-02","arxiv_id":"2402.01298","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-asr-robust-spoken-language","title":"Towards ASR Robust Spoken Language Understanding Through In-Context Learning With Word Confusion Networks","date":"2024-01-05","arxiv_id":"2401.02921","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-generalization-in-spoken","title":"Compositional Generalization in Spoken Language Understanding","date":"2023-12-25","arxiv_id":"2312.15815","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-context-aware-fine-tuning-of-self","title":"Generative Context-aware Fine-tuning of Self-supervised Speech Models","date":"2023-12-15","arxiv_id":"2312.09895","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-spoken-dialog-systems-in-ultra-low","title":"Creating Spoken Dialog Systems in Ultra-Low Resourced Settings","date":"2023-12-11","arxiv_id":"2312.06266","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-guiding-for-multi-intent-spoken-language","title":"Co-guiding for Multi-intent Spoken Language Understanding","date":"2023-11-22","arxiv_id":"2312.03716","repositories_listed":0,"syntology":null},{"url":null,"slug":"ml-lmcl-mutual-learning-and-large-margin","title":"ML-LMCL: Mutual Learning and Large-Margin Contrastive Learning for Improving ASR Robustness in Spoken Language Understanding","date":"2023-11-19","arxiv_id":"2311.11375","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-zero-shot-audio-to-intent","title":"Generalized zero-shot audio-to-intent classification","date":"2023-11-04","arxiv_id":"2311.02482","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-joint-language-modeling-for-speech","title":"Toward Joint Language Modeling for Speech Units and Text","date":"2023-10-12","arxiv_id":"2310.08715","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-spoken-language-understanding-via","title":"Few-Shot Spoken Language Understanding via Joint Speech-Text Models","date":"2023-10-09","arxiv_id":"2310.05919","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-end-to-end-speech-processing-by","title":"Improving End-to-End Speech Processing by Efficient Text Data Utilization with Latent Synthesis","date":"2023-10-09","arxiv_id":"2310.05374","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-contrastive-spoken-language","title":"Continual Contrastive Spoken Language Understanding","date":"2023-10-04","arxiv_id":"2310.02699","repositories_listed":0,"syntology":null},{"url":null,"slug":"i-2-kd-slu-an-intra-inter-knowledge","title":"I$^2$KD-SLU: An Intra-Inter Knowledge Distillation Framework for Zero-Shot Cross-Lingual Spoken Language Understanding","date":"2023-10-04","arxiv_id":"2310.02594","repositories_listed":0,"syntology":null},{"url":"/paper/universlu-universal-spoken-language","slug":"universlu-universal-spoken-language","title":"UniverSLU: Universal Spoken Language Understanding for Diverse Tasks with Natural Language Instructions","date":"2023-10-04","arxiv_id":"2310.02973","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-speech-recognition-translation-and","title":"Exploring Speech Recognition, Translation, and Understanding with Discrete Speech Units: A Comparative Study","date":"2023-09-27","arxiv_id":"2309.15800","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-speaker-diarization-using-semantic","title":"Improving Speaker Diarization using Semantic Information: Joint Pairwise Constraints Propagation","date":"2023-09-19","arxiv_id":"2309.10456","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-text-for-spoken-language","title":"Augmenting text for spoken language understanding with Large Language Models","date":"2023-09-17","arxiv_id":"2309.09390","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-large-language-models-for","title":"Leveraging Large Language Models for Exploiting ASR Uncertainty","date":"2023-09-09","arxiv_id":"2309.04842","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-the-blind-spots-in-spoken-language","title":"Addressing the Blind Spots in Spoken Language Processing","date":"2023-09-06","arxiv_id":"2309.06572","repositories_listed":0,"syntology":null},{"url":null,"slug":"lada-latent-dialogue-action-for-zero-shot","title":"LaDA: Latent Dialogue Action For Zero-shot Cross-lingual Neural Network Language Modeling","date":"2023-08-05","arxiv_id":"2308.02903","repositories_listed":0,"syntology":null},{"url":null,"slug":"modality-confidence-aware-training-for-robust","title":"Modality Confidence Aware Training for Robust End-to-End Spoken Language Understanding","date":"2023-07-22","arxiv_id":"2307.12134","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-pretrained-asr-and-lm-to-perform","title":"Integrating Pretrained ASR and LM to Perform Sequence Generation for Spoken Language Understanding","date":"2023-07-20","arxiv_id":"2307.11005","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-enrichment-towards-efficient-speech","title":"Semantic enrichment towards efficient speech representations","date":"2023-07-03","arxiv_id":"2307.01323","repositories_listed":0,"syntology":null},{"url":null,"slug":"multimodal-audio-textual-architecture-for-1","title":"Multimodal Audio-textual Architecture for Robust Spoken Language Understanding","date":"2023-06-12","arxiv_id":"2306.06819","repositories_listed":0,"syntology":null},{"url":null,"slug":"tensor-decomposition-for-minimization-of-e2e","title":"Tensor decomposition for minimization of E2E SLU model toward on-device processing","date":"2023-06-02","arxiv_id":"2306.01247","repositories_listed":0,"syntology":null},{"url":null,"slug":"inspecting-spoken-language-understanding-from","title":"Inspecting Spoken Language Understanding from Kids for Basic Math Learning at Home","date":"2023-06-01","arxiv_id":"2306.00482","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-textless-spoken-language","title":"Improving Textless Spoken Language Understanding with Discrete Units as Intermediate Target","date":"2023-05-29","arxiv_id":"2305.18096","repositories_listed":0,"syntology":null},{"url":null,"slug":"cif-pt-bridging-speech-and-text","title":"CIF-PT: Bridging Speech and Text Representations for Spoken Language Understanding via Continuous Integrate-and-Fire Pre-Training","date":"2023-05-27","arxiv_id":"2305.17499","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-level-knowledge-distillation-for-1","title":"Sequence-Level Knowledge Distillation for Class-Incremental End-to-End Spoken Language Understanding","date":"2023-05-23","arxiv_id":"2305.13899","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-chatgpt-detect-intent-evaluating-large","title":"Can ChatGPT Detect Intent? Evaluating Large Language Models for Spoken Language Understanding","date":"2023-05-22","arxiv_id":"2305.13512","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-speaker-related-information-in","title":"Exploring Speaker-Related Information in Spoken Language Understanding for Better Speaker Diarization","date":"2023-05-22","arxiv_id":"2305.12927","repositories_listed":0,"syntology":null},{"url":"/paper/fast-conformer-with-linearly-scalable","slug":"fast-conformer-with-linearly-scalable","title":"Fast Conformer with Linearly Scalable Attention for Efficient Speech Recognition","date":"2023-05-08","arxiv_id":"2305.05084","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-spoken-language-understanding-4","title":"End-to-end spoken language understanding using joint CTC loss and self-supervised, pretrained acoustic encoders","date":"2023-05-04","arxiv_id":"2305.02937","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-the-integration-of-pipeline-and","title":"A Study on the Integration of Pipeline and E2E SLU systems for Spoken Semantic Parsing toward STOP Quality Challenge","date":"2023-05-02","arxiv_id":"2305.01620","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-pipeline-system-of-asr-and-nlu-with-mlm","title":"The Pipeline System of ASR and NLU with MLM-based Data Augmentation toward STOP Low-resource Challenge","date":"2023-05-02","arxiv_id":"2305.01194","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-modelling-of-spoken-language","title":"Joint Modelling of Spoken Language Understanding Tasks with Integrated Dialog History","date":"2023-05-01","arxiv_id":"2305.00926","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-autoregressive-end-to-end-approaches-for","title":"Non-autoregressive End-to-end Approaches for Joint Automatic Speech Recognition and Spoken Language Understanding","date":"2023-04-21","arxiv_id":"2304.10869","repositories_listed":0,"syntology":null}],"record_sha256":"fa9a33acf01125ab8f9d08b7a867ed8d04e530e6674a11e5d797321ecea42a54","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}