{"url":"/task/open-domain-dialog","name":"Open-Domain Dialog","slug":"open-domain-dialog","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":60,"papers_with_code":32,"benchmarks":1,"benchmark_tables_in_archive":1,"benchmark_tables_shown":1,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":13,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/open-domain-dialog-on-kilt-wizard-of","slug":"open-domain-dialog-on-kilt-wizard-of","dataset":"KILT: Wizard of Wikipedia","dataset_url":"/dataset/kilt","rows_in_archive":21,"metrics":["KILT-RL","R-Prec","Recall@5","ROUGE-L","F1","KILT-F1"],"first_row_in_archive_order":{"model":"Hindsight","paper_title":null,"paper_url":null,"paper_date":"","arxiv_id":null,"code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/reddit","name":"Reddit","full_name":"","num_papers_in_archive":699},{"url":"/dataset/kilt","name":"KILT","full_name":"KILT Benchmark","num_papers_in_archive":117},{"url":"/dataset/multidoc2dial","name":"MultiDoc2Dial","full_name":"MultiDoc2Dial: Modeling Dialogues Grounded in Multiple Documents","num_papers_in_archive":26},{"url":"/dataset/mmdialog","name":"MMDialog","full_name":"","num_papers_in_archive":17},{"url":"/dataset/cped","name":"CPED","full_name":"Chinese Personalized and Emotional Dialogue","num_papers_in_archive":15},{"url":"/dataset/delemon","name":"DuLeMon","full_name":"Baidu Long-term Memory Conversation","num_papers_in_archive":14},{"url":"/dataset/prosocialdialog","name":"ProsocialDialog","full_name":"","num_papers_in_archive":13},{"url":"/dataset/reddit-conversation-corpus","name":"Reddit Conversation Corpus","full_name":"","num_papers_in_archive":6},{"url":"/dataset/clovacall","name":"ClovaCall","full_name":null,"num_papers_in_archive":5},{"url":"/dataset/mmchat","name":"MMChat","full_name":"","num_papers_in_archive":5},{"url":"/dataset/diamante","name":"Diamante","full_name":"","num_papers_in_archive":4},{"url":"/dataset/cpsycound","name":"CPsyCounD","full_name":"","num_papers_in_archive":3},{"url":"/dataset/cpsycoune","name":"CPsyCounE","full_name":"","num_papers_in_archive":2}],"subtasks":[{"url":"/task/dialogue-evaluation","name":"Dialogue Evaluation"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":32,"tagged_in_all":60,"items":[{"url":"/paper/prophetnet-x-large-scale-pre-training-models","title":"ProphetNet-X: Large-Scale Pre-training Models for English, Chinese, Multi-lingual, Dialog, and Code Generation","date":"2021-04-16","arxiv_id":"2104.08006","repositories_listed":3,"syntology":null},{"url":"/paper/kilt-a-benchmark-for-knowledge-intensive","title":"KILT: a Benchmark for Knowledge Intensive Language Tasks","date":"2020-09-04","arxiv_id":"2009.02252","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/hurdles-to-progress-in-long-form-question","title":"Hurdles to Progress in Long-form Question Answering","date":"2021-03-10","arxiv_id":"2103.06332","repositories_listed":2,"syntology":null},{"url":"/paper/dialogue-response-ranking-training-with-large","title":"Dialogue Response Ranking Training with Large-Scale Human Feedback Data","date":"2020-09-15","arxiv_id":"2009.06978","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/unsupervised-evaluation-of-interactive-dialog","title":"Unsupervised Evaluation of Interactive Dialog with DialoGPT","date":"2020-06-23","arxiv_id":"2006.12719","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/investigating-evaluation-of-open-domain","title":"Investigating Evaluation of Open-Domain Dialogue Systems With Human Generated Multiple References","date":"2019-07-24","arxiv_id":"1907.10568","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/approximating-interactive-human-evaluation","title":"Approximating Interactive Human Evaluation with Self-Play for Open-Domain Dialog Systems","date":"2019-06-21","arxiv_id":"1906.09308","repositories_listed":2,"syntology":null},{"url":"/paper/an-empirical-study-on-context-length-for-open","title":"An Empirical Study on Context Length for Open-Domain Dialog Generation","date":"2024-08-31","arxiv_id":"2409.00315","repositories_listed":1,"syntology":null},{"url":"/paper/dior-cvae-diffusion-priors-in-variational","title":"Dior-CVAE: Pre-trained Language Models and Diffusion Priors for Variational Dialog Generation","date":"2023-05-24","arxiv_id":"2305.15025","repositories_listed":1,"syntology":null},{"url":"/paper/open-domain-dialog-evaluation-using-follow","title":"Open-Domain Dialog Evaluation using Follow-Ups Likelihood","date":"2022-09-12","arxiv_id":"2209.05185","repositories_listed":1,"syntology":null},{"url":"/paper/re2g-retrieve-rerank-generate-2","title":"Re2G: Retrieve, Rerank, Generate","date":"2022-07-13","arxiv_id":"2207.06300","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/godel-large-scale-pre-training-for-goal","title":"GODEL: Large-Scale Pre-Training for Goal-Directed Dialog","date":"2022-06-22","arxiv_id":"2206.11309","repositories_listed":1,"syntology":null},{"url":"/paper/cped-a-large-scale-chinese-personalized-and-1","title":"CPED: A Large-Scale Chinese Personalized and Emotional Dialogue Dataset for Conversational AI","date":"2022-05-29","arxiv_id":"2205.14727","repositories_listed":1,"syntology":null},{"url":"/paper/improving-zero-and-few-shot-generalization-in","title":"InstructDial: Improving Zero and Few-shot Generalization in Dialogue through Instruction Tuning","date":"2022-05-25","arxiv_id":"2205.12673","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/what-is-wrong-with-you-leveraging-user","title":"What is wrong with you?: Leveraging User Sentiment for Automatic Dialog Evaluation","date":"2022-03-25","arxiv_id":"2203.13927","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-robustness-of-dialog-models-to","title":"Investigating Robustness of Dialog Models to Popular Figurative Language Constructs","date":"2021-10-01","arxiv_id":"2110.00687","repositories_listed":1,"syntology":null},{"url":"/paper/gensf-simultaneous-adaptation-of-generative","title":"GenSF: Simultaneous Adaptation of Generative Pre-trained Models and Slot Filling","date":"2021-06-13","arxiv_id":"2106.07055","repositories_listed":1,"syntology":null},{"url":"/paper/improving-automated-evaluation-of-open-domain","title":"Improving Automated Evaluation of Open Domain Dialog via Diverse Reference Augmentation","date":"2021-06-05","arxiv_id":"2106.02833","repositories_listed":1,"syntology":null},{"url":"/paper/herald-an-annotation-efficient-method-to","title":"HERALD: An Annotation Efficient Method to Detect User Disengagement in Social Conversations","date":"2021-06-01","arxiv_id":"2106.00162","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/comae-a-multi-factor-hierarchical-framework","title":"CoMAE: A Multi-factor Hierarchical Framework for Empathetic Response Generation","date":"2021-05-18","arxiv_id":"2105.08316","repositories_listed":1,"syntology":null},{"url":"/paper/are-neural-open-domain-dialog-systems-robust","title":"Are Neural Open-Domain Dialog Systems Robust to Speech Recognition Errors in the Dialog History? An Empirical Study","date":"2020-08-18","arxiv_id":"2008.07683","repositories_listed":1,"syntology":null},{"url":"/paper/probing-neural-dialog-models-for","title":"Probing Neural Dialog Models for Conversational Understanding","date":"2020-06-07","arxiv_id":"2006.08331","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-user-self-reported-likert-scale","title":"Beyond User Self-Reported Likert Scale Ratings: A Comparison Model for Automatic Dialog Evaluation","date":"2020-05-21","arxiv_id":"2005.10716","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/usr-an-unsupervised-and-reference-free","title":"USR: An Unsupervised and Reference Free Evaluation Metric for Dialog Generation","date":"2020-05-01","arxiv_id":"2005.00456","repositories_listed":1,"syntology":null},{"url":"/paper/clovacall-korean-goal-oriented-dialog-speech","title":"ClovaCall: Korean Goal-Oriented Dialog Speech Corpus for Automatic Speech Recognition of Contact Centers","date":"2020-04-20","arxiv_id":"2004.09367","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-for-open","title":"Hierarchical Reinforcement Learning for Open-Domain Dialog","date":"2019-09-17","arxiv_id":"1909.07547","repositories_listed":1,"syntology":null},{"url":"/paper/190807816","title":"A Multi-Turn Emotionally Engaging Dialog Model","date":"2019-08-15","arxiv_id":"1908.07816","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-transfer-learning-for-natural","title":"Large-Scale Transfer Learning for Natural Language Generation","date":"2019-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/way-off-policy-batch-deep-reinforcement","title":"Way Off-Policy Batch Deep Reinforcement Learning of Implicit Human Preferences in Dialog","date":"2019-06-30","arxiv_id":"1907.00456","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/evaluating-coherence-in-dialogue-systems","title":"Evaluating Coherence in Dialogue Systems using Entailment","date":"2019-04-06","arxiv_id":"1904.03371","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":1}}],"syntology_records":10,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}