{"url":"/task/dialogue-state-tracking","name":"Dialogue State Tracking","slug":"dialogue-state-tracking","description_markdown":"Dialogue state tacking consists of determining at each turn of a dialogue the\nfull representation of what the user wants at that point in the dialogue,\nwhich contains a goal constraint, a set of requested slots, and the user's dialogue act.","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"},{"name":"Speech","url":"/area/speech"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":300,"papers_with_code":138,"benchmarks":7,"benchmark_tables_in_archive":7,"benchmark_tables_shown":7,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":13,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/dialogue-state-tracking-on-wizard-of-oz","slug":"dialogue-state-tracking-on-wizard-of-oz","dataset":"Wizard-of-Oz","dataset_url":"/dataset/wizard-of-oz","rows_in_archive":10,"metrics":["Joint","Request"],"first_row_in_archive_order":{"model":"AG-DST","paper_title":"Amendable Generation for Dialogue State Tracking","paper_url":"/paper/amendable-generation-for-dialogue-state","paper_date":"2021-10-29","arxiv_id":"2110.15659","code_links":[{"title":"PaddlePaddle/Knover","url":"https://github.com/PaddlePaddle/Knover/tree/develop/projects/AG-DST"}],"syntology":null}},{"leaderboard":"/sota/dialogue-state-tracking-on-cosql","slug":"dialogue-state-tracking-on-cosql","dataset":"CoSQL","dataset_url":"/dataset/cosql","rows_in_archive":9,"metrics":["question match accuracy","interaction match accuracy"],"first_row_in_archive_order":{"model":"RASAT+PICARD","paper_title":"RASAT: Integrating Relational Structures into Pretrained Seq2Seq Model for Text-to-SQL","paper_url":"/paper/rasat-integrating-relational-structures-into","paper_date":"2022-05-14","arxiv_id":"2205.06983","code_links":[{"title":"lumia-group/rasat","url":"https://github.com/lumia-group/rasat"}],"syntology":{"n":9,"n_ran":0,"n_unverified":9,"n_pointer_only":0}}},{"leaderboard":"/sota/dialogue-state-tracking-on-second-dialogue","slug":"dialogue-state-tracking-on-second-dialogue","dataset":"Second dialogue state tracking challenge","dataset_url":"/dataset/dialogue-state-tracking-challenge","rows_in_archive":7,"metrics":["Joint","Area","Food","Price","Request"],"first_row_in_archive_order":{"model":"Seq2Seq-DU-w/oSchema","paper_title":"A Sequence-to-Sequence Approach to Dialogue State Tracking","paper_url":"/paper/a-sequence-to-sequence-approach-to-dialogue","paper_date":"2020-11-18","arxiv_id":"2011.09553","code_links":[{"title":"sweetalyssum/Seq2Seq-DU","url":"https://github.com/sweetalyssum/Seq2Seq-DU"}],"syntology":null}},{"leaderboard":"/sota/dialogue-state-tracking-on-simmc2-0","slug":"dialogue-state-tracking-on-simmc2-0","dataset":"SIMMC2.0","dataset_url":"/dataset/simmc2-0","rows_in_archive":5,"metrics":["Act F1","Slot F1"],"first_row_in_archive_order":{"model":"PaCE","paper_title":"PaCE: Unified Multi-modal Dialogue Pre-training with Progressive and Compositional Experts","paper_url":"/paper/pace-unified-multi-modal-dialogue-pre","paper_date":"2023-05-24","arxiv_id":"2305.14839","code_links":[{"title":"AlibabaResearch/DAMO-ConvAI","url":"https://github.com/AlibabaResearch/DAMO-ConvAI/tree/main/pace"}],"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/dialogue-state-tracking-on-multiwoz-2-1","slug":"dialogue-state-tracking-on-multiwoz-2-1","dataset":"MULTIWOZ 2.1","dataset_url":"/dataset/multiwoz","rows_in_archive":4,"metrics":["Joint Acc","MultiWOZ (Joint Goal Acc)"],"first_row_in_archive_order":{"model":"DeepStruct multi-task w/ finetune","paper_title":"DeepStruct: Pretraining of Language Models for Structure Prediction","paper_url":"/paper/deepstruct-pretraining-of-language-models-for-1","paper_date":"2022-05-21","arxiv_id":"2205.10475","code_links":[{"title":"cgraywang/deepstruct","url":"https://github.com/cgraywang/deepstruct"}],"syntology":{"n":13,"n_ran":7,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/dialogue-state-tracking-on-mmconv","slug":"dialogue-state-tracking-on-mmconv","dataset":"MMConv","dataset_url":"/dataset/mmconv","rows_in_archive":2,"metrics":["Categorical Accuracy","Non-Categorical Accuracy","Overall"],"first_row_in_archive_order":{"model":"PaCE","paper_title":"PaCE: Unified Multi-modal Dialogue Pre-training with Progressive and Compositional Experts","paper_url":"/paper/pace-unified-multi-modal-dialogue-pre","paper_date":"2023-05-24","arxiv_id":"2305.14839","code_links":[{"title":"AlibabaResearch/DAMO-ConvAI","url":"https://github.com/AlibabaResearch/DAMO-ConvAI/tree/main/pace"}],"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/dialogue-state-tracking-on-multiwoz-2-2","slug":"dialogue-state-tracking-on-multiwoz-2-2","dataset":"MULTIWOZ 2.2","dataset_url":"/dataset/multiwoz","rows_in_archive":2,"metrics":["MultiWOZ (Joint Goal Acc)"],"first_row_in_archive_order":{"model":"SGP-DST (base)","paper_title":"Dialogue State Tracking with a Language Model using Schema-Driven Prompting","paper_url":"/paper/dialogue-state-tracking-with-a-language-model","paper_date":"2021-09-15","arxiv_id":"2109.07506","code_links":[{"title":"chiahsuan156/dst-as-prompting","url":"https://github.com/chiahsuan156/dst-as-prompting"}],"syntology":null}}],"datasets":[{"url":"/dataset/multiwoz","name":"MultiWOZ","full_name":"Multi-domain Wizard-of-Oz","num_papers_in_archive":328},{"url":"/dataset/sgd","name":"SGD","full_name":"Schema-Guided Dialogue","num_papers_in_archive":186},{"url":"/dataset/cosql","name":"CoSQL","full_name":"Conversational Text-to-SQL Challenge","num_papers_in_archive":45},{"url":"/dataset/dialogue-state-tracking-challenge","name":"Dialogue State Tracking Challenge","full_name":"Dialogue State Tracking Challenge","num_papers_in_archive":33},{"url":"/dataset/wizard-of-oz","name":"Wizard-of-Oz","full_name":"","num_papers_in_archive":28},{"url":"/dataset/crosswoz","name":"CrossWOZ","full_name":"CrossWOZ","num_papers_in_archive":25},{"url":"/dataset/klue","name":"KLUE","full_name":"Korean Language Understanding Evaluation","num_papers_in_archive":21},{"url":"/dataset/taskmaster-1","name":"Taskmaster-1","full_name":"Taskmaster-1","num_papers_in_archive":19},{"url":"/dataset/simmc2-0","name":"SIMMC2.0","full_name":"","num_papers_in_archive":13},{"url":"/dataset/risawoz","name":"RiSAWOZ","full_name":null,"num_papers_in_archive":5},{"url":"/dataset/mmconv","name":"MMConv","full_name":"","num_papers_in_archive":3},{"url":"/dataset/diaforge-utc-r-0725","name":"diaforge-utc-r-0725","full_name":"DiaFORGE UTC: Unified Tool-Calling Conversations Dataset","num_papers_in_archive":1},{"url":"/dataset/indirectrequests","name":"IndirectRequests","full_name":"IndirectRequests","num_papers_in_archive":0}],"subtasks":[],"parent_tasks":[{"url":"/task/dialogue","name":"Dialogue"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":138,"tagged_in_all":300,"items":[{"url":"/paper/language-models-are-unsupervised-multitask","title":"Language Models are Unsupervised Multitask Learners","date":"2019-02-14","arxiv_id":null,"repositories_listed":21,"syntology":null},{"url":"/paper/multiwoz-21-multi-domain-dialogue-state","title":"MultiWOZ 2.1: A Consolidated Multi-Domain Dialogue Dataset with State Corrections and State Tracking Baselines","date":"2019-07-02","arxiv_id":"1907.01669","repositories_listed":6,"syntology":{"n":23,"n_ran":2,"n_unverified":21,"n_pointer_only":0}},{"url":"/paper/klue-korean-language-understanding-evaluation","title":"KLUE: Korean Language Understanding Evaluation","date":"2021-05-20","arxiv_id":"2105.09680","repositories_listed":4,"syntology":null},{"url":"/paper/towards-scalable-multi-domain-conversational","title":"Towards Scalable Multi-domain Conversational Agents: The Schema-Guided Dialogue Dataset","date":"2019-09-12","arxiv_id":"1909.05855","repositories_listed":4,"syntology":null},{"url":"/paper/picard-parsing-incrementally-for-constrained","title":"PICARD: Parsing Incrementally for Constrained Auto-Regressive Decoding from Language Models","date":"2021-09-10","arxiv_id":"2109.05093","repositories_listed":3,"syntology":{"n":7,"n_ran":4,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/multiwoz-2-3-a-multi-domain-task-oriented","title":"MultiWOZ 2.3: A multi-domain task-oriented dialogue dataset enhanced with annotation corrections and co-reference annotation","date":"2020-10-12","arxiv_id":"2010.05594","repositories_listed":3,"syntology":null},{"url":"/paper/efficient-dialogue-state-tracking-by","title":"Efficient Dialogue State Tracking by Selectively Overwriting Memory","date":"2019-11-10","arxiv_id":"1911.03906","repositories_listed":3,"syntology":null},{"url":"/paper/cosql-a-conversational-text-to-sql-challenge","title":"CoSQL: A Conversational Text-to-SQL Challenge Towards Cross-Domain Natural Language Interfaces to Databases","date":"2019-09-11","arxiv_id":"1909.05378","repositories_listed":3,"syntology":{"n":11,"n_ran":1,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/editing-based-sql-query-generation-for-cross","title":"Editing-Based SQL Query Generation for Cross-Domain Context-Dependent Questions","date":"2019-09-02","arxiv_id":"1909.00786","repositories_listed":3,"syntology":null},{"url":"/paper/checkdst-measuring-real-world-generalization","title":"Know Thy Strengths: Comprehensive Dialogue State Tracking Diagnostics","date":"2021-12-15","arxiv_id":"2112.08321","repositories_listed":2,"syntology":null},{"url":"/paper/few-shot-bot-prompt-based-learning-for","title":"Few-Shot Bot: Prompt-Based Learning for Dialogue Systems","date":"2021-10-15","arxiv_id":"2110.08118","repositories_listed":2,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/multi-task-pre-training-for-plug-and-play","title":"Multi-Task Pre-Training for Plug-and-Play Task-Oriented Dialogue System","date":"2021-09-29","arxiv_id":"2109.14739","repositories_listed":2,"syntology":{"n":10,"n_ran":0,"n_unverified":10,"n_pointer_only":0}},{"url":"/paper/leveraging-slot-descriptions-for-zero-shot","title":"Leveraging Slot Descriptions for Zero-Shot Cross-Domain Dialogue State Tracking","date":"2021-05-10","arxiv_id":"2105.04222","repositories_listed":2,"syntology":null},{"url":"/paper/structured-prediction-as-translation-between-1","title":"Structured Prediction as Translation between Augmented Natural Languages","date":"2021-01-14","arxiv_id":"2101.05779","repositories_listed":2,"syntology":null},{"url":"/paper/dynamic-hybrid-relation-network-for-cross","title":"Dynamic Hybrid Relation Network for Cross-Domain Context-Dependent Semantic Parsing","date":"2021-01-05","arxiv_id":"2101.01686","repositories_listed":2,"syntology":null},{"url":"/paper/coco-controllable-counterfactuals-for-1","title":"CoCo: Controllable Counterfactuals for Evaluating Dialogue State Trackers","date":"2020-10-24","arxiv_id":"2010.12850","repositories_listed":2,"syntology":{"n":10,"n_ran":9,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/multi-domain-dialogue-state-tracking-a-purely","title":"Jointly Optimizing State Operation Prediction and Value Generation for Dialogue State Tracking","date":"2020-10-24","arxiv_id":"2010.14061","repositories_listed":2,"syntology":null},{"url":"/paper/crosswoz-a-large-scale-chinese-cross-domain","title":"CrossWOZ: A Large-Scale Chinese Cross-Domain Task-Oriented Dialogue Dataset","date":"2020-02-27","arxiv_id":"2002.11893","repositories_listed":2,"syntology":null},{"url":"/paper/schema-guided-dialogue-state-tracking-task-at","title":"Schema-Guided Dialogue State Tracking Task at DSTC8","date":"2020-02-02","arxiv_id":"2002.01359","repositories_listed":2,"syntology":null},{"url":"/paper/multi-domain-dialogue-state-tracking-as","title":"Multi-domain Dialogue State Tracking as Dynamic Knowledge Graph Enhanced Question Answering","date":"2019-11-07","arxiv_id":"1911.06192","repositories_listed":2,"syntology":null},{"url":"/paper/transferable-multi-domain-state-generator-for","title":"Transferable Multi-Domain State Generator for Task-Oriented Dialogue Systems","date":"2019-05-21","arxiv_id":"1905.08743","repositories_listed":2,"syntology":null},{"url":"/paper/explicit-state-tracking-with-semi-supervision","title":"Explicit State Tracking with Semi-Supervision for Neural Dialogue Generation","date":"2018-08-31","arxiv_id":"1808.10596","repositories_listed":2,"syntology":null},{"url":"/paper/global-locally-self-attentive-dialogue-state","title":"Global-Locally Self-Attentive Dialogue State Tracker","date":"2018-05-19","arxiv_id":"1805.09655","repositories_listed":2,"syntology":{"n":8,"n_ran":0,"n_unverified":8,"n_pointer_only":3}},{"url":"/paper/semantic-specialisation-of-distributional","title":"Semantic Specialisation of Distributional Word Vector Spaces using Monolingual and Cross-Lingual Constraints","date":"2017-06-01","arxiv_id":"1706.00374","repositories_listed":2,"syntology":null},{"url":"/paper/counter-fitting-word-vectors-to-linguistic","title":"Counter-fitting Word Vectors to Linguistic Constraints","date":"2016-03-02","arxiv_id":"1603.00892","repositories_listed":2,"syntology":null},{"url":"/paper/towards-preventing-overreliance-on-task","title":"Know Your Mistakes: Towards Preventing Overreliance on Task-Oriented Conversational AI Through Accountability Modeling","date":"2025-01-17","arxiv_id":"2501.10316","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-ontology-in-dialogue-state-tracking","title":"Beyond Ontology in Dialogue State Tracking for Goal-Oriented Chatbot","date":"2024-10-30","arxiv_id":"2410.22767","repositories_listed":1,"syntology":null},{"url":"/paper/a-zero-shot-open-vocabulary-pipeline-for","title":"A Zero-Shot Open-Vocabulary Pipeline for Dialogue Understanding","date":"2024-09-24","arxiv_id":"2409.15861","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-estimation-for-llm-based-dialogue","title":"Confidence Estimation for LLM-Based Dialogue State Tracking","date":"2024-09-15","arxiv_id":"2409.09629","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_unverified":2,"n_pointer_only":12}},{"url":"/paper/tasl-continual-dialog-state-tracking-via-task","title":"TaSL: Continual Dialog State Tracking via Task Skill Localization and Consolidation","date":"2024-08-19","arxiv_id":"2408.09857","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}}],"syntology_records":9,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}