{"url":"/task/dialogue-evaluation","name":"Dialogue Evaluation","slug":"dialogue-evaluation","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":97,"papers_with_code":61,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":8,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/dialogue-evaluation-on-usr-topicalchat","slug":"dialogue-evaluation-on-usr-topicalchat","dataset":"USR-TopicalChat","dataset_url":"/dataset/usr-topicalchat","rows_in_archive":6,"metrics":["Spearman Correlation","Pearson Correlation"],"first_row_in_archive_order":{"model":"MDD-Eval","paper_title":"MDD-Eval: Self-Training on Augmented Data for Multi-Domain Dialogue Evaluation","paper_url":"/paper/mdd-eval-self-training-on-augmented-data-for","paper_date":"2021-12-14","arxiv_id":"2112.07194","code_links":[{"title":"e0397123/mdd-eval","url":"https://github.com/e0397123/mdd-eval"}],"syntology":null}},{"leaderboard":"/sota/dialogue-evaluation-on-usr-personachat","slug":"dialogue-evaluation-on-usr-personachat","dataset":"USR-PersonaChat","dataset_url":"/dataset/usr-personachat","rows_in_archive":5,"metrics":["Spearman Correlation","Pearson Correlation"],"first_row_in_archive_order":{"model":"Lin-Reg (all)","paper_title":"Proxy Indicators for the Quality of Open-domain Dialogues","paper_url":"/paper/proxy-indicators-for-the-quality-of-open","paper_date":"","arxiv_id":null,"code_links":[{"title":"smartdataanalytics/proxy_indicators","url":"https://github.com/smartdataanalytics/proxy_indicators"}],"syntology":null}}],"datasets":[{"url":"/dataset/reddit","name":"Reddit","full_name":"","num_papers_in_archive":699},{"url":"/dataset/faithdial","name":"FaithDial","full_name":"","num_papers_in_archive":16},{"url":"/dataset/usr-personachat","name":"USR-PersonaChat","full_name":"","num_papers_in_archive":7},{"url":"/dataset/usr-topicalchat","name":"USR-TopicalChat","full_name":"","num_papers_in_archive":7},{"url":"/dataset/cpsycoune","name":"CPsyCounE","full_name":"","num_papers_in_archive":2},{"url":"/dataset/diaforge-utc-r-0725","name":"diaforge-utc-r-0725","full_name":"DiaFORGE UTC: Unified Tool-Calling Conversations Dataset","num_papers_in_archive":1},{"url":"/dataset/reddit-engagement-dataset","name":"Reddit Engagement Dataset","full_name":"","num_papers_in_archive":1},{"url":"/dataset/saga","name":"SaGA","full_name":"The Bielefeld Speech and Gesture Alignment Corpus (SaGA)","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/open-domain-dialog","name":"Open-Domain Dialog"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":61,"tagged_in_all":97,"items":[{"url":"/paper/adversarial-learning-for-neural-dialogue","title":"Adversarial Learning for Neural Dialogue Generation","date":"2017-01-23","arxiv_id":"1701.06547","repositories_listed":8,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/don-t-forget-your-abc-s-evaluating-the-state","title":"Don't Forget Your ABC's: Evaluating the State-of-the-Art in Chat-Oriented Dialogue Systems","date":"2022-12-18","arxiv_id":"2212.09180","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/fined-eval-fine-grained-automatic-dialogue","title":"FineD-Eval: Fine-grained Automatic Dialogue-Level Evaluation","date":"2022-10-25","arxiv_id":"2210.13832","repositories_listed":2,"syntology":null},{"url":"/paper/automatic-evaluation-and-moderation-of-open","title":"Automatic Evaluation and Moderation of Open-domain Dialogue Systems","date":"2021-11-03","arxiv_id":"2111.02110","repositories_listed":2,"syntology":null},{"url":"/paper/unsupervised-evaluation-of-interactive-dialog","title":"Unsupervised Evaluation of Interactive Dialog with DialoGPT","date":"2020-06-23","arxiv_id":"2006.12719","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/predictive-engagement-an-efficient-metric-for","title":"Predictive Engagement: An Efficient Metric For Automatic Evaluation of Open-Domain Dialogue Systems","date":"2019-11-04","arxiv_id":"1911.01456","repositories_listed":2,"syntology":null},{"url":"/paper/investigating-evaluation-of-open-domain","title":"Investigating Evaluation of Open-Domain Dialogue Systems With Human Generated Multiple References","date":"2019-07-24","arxiv_id":"1907.10568","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/approximating-interactive-human-evaluation","title":"Approximating Interactive Human Evaluation with Self-Play for Open-Domain Dialog Systems","date":"2019-06-21","arxiv_id":"1906.09308","repositories_listed":2,"syntology":null},{"url":"/paper/medal-a-framework-for-benchmarking-llms-as","title":"MEDAL: A Framework for Benchmarking LLMs as Multilingual Open-Domain Chatbots and Dialogue Evaluators","date":"2025-05-28","arxiv_id":"2505.22777","repositories_listed":1,"syntology":null},{"url":"/paper/methods-for-recognizing-nested-terms","title":"Methods for Recognizing Nested Terms","date":"2025-04-22","arxiv_id":"2504.16007","repositories_listed":1,"syntology":null},{"url":"/paper/ruopinionne-2024-extraction-of-opinion-tuples","title":"RuOpinionNE-2024: Extraction of Opinion Tuples from Russian News Texts","date":"2025-04-09","arxiv_id":"2504.06947","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-single-turn-a-survey-on-multi-turn","title":"Beyond Single-Turn: A Survey on Multi-Turn Interactions with Large Language Models","date":"2025-04-07","arxiv_id":"2504.04717","repositories_listed":1,"syntology":null},{"url":"/paper/bok-introducing-bag-of-keywords-loss-for","title":"BoK: Introducing Bag-of-Keywords Loss for Interpretable Dialogue Response Generation","date":"2025-01-17","arxiv_id":"2501.10328","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-the-robustness-of-reference-free","title":"Measuring the Robustness of Reference-Free Dialogue Evaluation Systems","date":"2025-01-12","arxiv_id":"2501.06728","repositories_listed":1,"syntology":null},{"url":"/paper/soda-eval-open-domain-dialogue-evaluation-in","title":"Soda-Eval: Open-Domain Dialogue Evaluation in the age of LLMs","date":"2024-08-20","arxiv_id":"2408.10902","repositories_listed":1,"syntology":null},{"url":"/paper/ecoh-turn-level-coherence-evaluation-for","title":"ECoh: Turn-level Coherence Evaluation for Multilingual Dialogues","date":"2024-07-16","arxiv_id":"2407.11660","repositories_listed":1,"syntology":null},{"url":"/paper/slide-a-framework-integrating-small-and-large","title":"SLIDE: A Framework Integrating Small and Large Language Models for Open-Domain Dialogues Evaluation","date":"2024-05-24","arxiv_id":"2405.15924","repositories_listed":1,"syntology":null},{"url":"/paper/structured-information-matters-incorporating","title":"Emphasising Structured Information: Integrating Abstract Meaning Representation into LLMs for Enhanced Open-Domain Dialogue Evaluation","date":"2024-04-01","arxiv_id":"2404.01129","repositories_listed":1,"syntology":null},{"url":"/paper/paireval-open-domain-dialogue-evaluation-with","title":"PairEval: Open-domain Dialogue Evaluation with Pairwise Comparison","date":"2024-04-01","arxiv_id":"2404.01015","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-analysis-of-the-effectiveness","title":"A Comprehensive Analysis of the Effectiveness of Large Language Models as Automatic Dialogue Evaluators","date":"2023-12-24","arxiv_id":"2312.15407","repositories_listed":1,"syntology":null},{"url":"/paper/dialogbench-evaluating-llms-as-human-like","title":"DialogBench: Evaluating LLMs as Human-like Dialogue Systems","date":"2023-11-03","arxiv_id":"2311.01677","repositories_listed":1,"syntology":null},{"url":"/paper/xdial-eval-a-multilingual-open-domain","title":"xDial-Eval: A Multilingual Open-Domain Dialogue Evaluation Benchmark","date":"2023-10-13","arxiv_id":"2310.08958","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-impact-of-human-evaluator-group","title":"Exploring the Impact of Human Evaluator Group on Chat-Oriented Dialogue Evaluation","date":"2023-09-14","arxiv_id":"2309.07998","repositories_listed":1,"syntology":null},{"url":"/paper/simple-llm-prompting-is-state-of-the-art-for","title":"Simple LLM Prompting is State-of-the-Art for Robust and Multilingual Dialogue Evaluation","date":"2023-08-31","arxiv_id":"2308.16797","repositories_listed":1,"syntology":null},{"url":"/paper/towards-multilingual-automatic-dialogue","title":"Towards Multilingual Automatic Dialogue Evaluation","date":"2023-08-31","arxiv_id":"2308.16795","repositories_listed":1,"syntology":null},{"url":"/paper/c-pmi-conditional-pointwise-mutual","title":"C-PMI: Conditional Pointwise Mutual Information for Turn-level Dialogue Evaluation","date":"2023-06-27","arxiv_id":"2306.15245","repositories_listed":1,"syntology":null},{"url":"/paper/density-open-domain-dialogue-evaluation","title":"DEnsity: Open-domain Dialogue Evaluation Metric using Density Estimation","date":"2023-05-08","arxiv_id":"2305.04720","repositories_listed":1,"syntology":null},{"url":"/paper/glm-dialog-noise-tolerant-pre-training-for","title":"GLM-Dialog: Noise-tolerant Pre-training for Knowledge-grounded Dialogue Generation","date":"2023-02-28","arxiv_id":"2302.14401","repositories_listed":1,"syntology":null},{"url":"/paper/self-eval-self-supervised-fine-grained","title":"SelF-Eval: Self-supervised Fine-grained Dialogue Evaluation","date":"2022-08-17","arxiv_id":"2208.08094","repositories_listed":1,"syntology":null},{"url":"/paper/findings-of-the-the-ruatd-shared-task-2022-on","title":"Findings of the The RuATD Shared Task 2022 on Artificial Text Detection in Russian","date":"2022-06-03","arxiv_id":"2206.01583","repositories_listed":1,"syntology":null}],"syntology_records":4,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}