{"url":"/method/enigma","slug":"enigma","name":"ENIGMA","full_name":"ENIGMA","full_name_withheld":false,"description_markdown":"**ENIGMA** is an evaluation framework for dialog systems based on Pearson and Spearman's rank correlations between the estimated rewards and the true rewards.  ENIGMA only requires a handful of pre-collected experience data, and therefore does not involve human interaction with the target policy during the evaluation, making automatic evaluations feasible. More importantly, ENIGMA is model-free and agnostic to the behavior policies for collecting the experience data (see details in Section 2), which significantly alleviates the technical difficulties of modeling complex dialogue environments and human behaviors.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Towards Automatic Evaluation of Dialog Systems: A Model-Free Off-Policy Evaluation Approach","paper":"/paper/towards-automatic-evaluation-of-dialog","first_author":"Haoming Jiang","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/towards-automatic-evaluation-of-dialog"},"source":{"url":"https://arxiv.org/abs/2102.10242v3","title":"Towards Automatic Evaluation of Dialog Systems: A Model-Free Off-Policy Evaluation Approach","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Natural Language Processing","area_id":"natural-language-processing","collection":"Dialog System Evaluation","url":"/methods/category/dialog-system-evaluation","pwc_aliases":[]}],"n_papers_tagged":7,"archive_num_papers":7,"papers_newest_first":[{"paper":null,"title":"MizAR 60 for Mizar 50","date":"2023-03-12","arxiv_id":"2303.06686","n_code_links":0,"syntology":null},{"paper":null,"title":"Multi-site benchmark classification of major depressive disorder using machine learning on cortical and subcortical measures","date":"2022-06-16","arxiv_id":"2206.08122","n_code_links":0,"syntology":null},{"paper":"/paper/the-isabelle-enigma","title":"The Isabelle ENIGMA","date":"2022-05-04","arxiv_id":"2205.01981","n_code_links":1,"syntology":null},{"paper":"/paper/learning-theorem-proving-components","title":"Learning Theorem Proving Components","date":"2021-07-21","arxiv_id":"2107.10034","n_code_links":1,"syntology":null},{"paper":"/paper/fast-and-slow-enigmas-and-parental-guidance","title":"Fast and Slow Enigmas and Parental Guidance","date":"2021-07-14","arxiv_id":"2107.06750","n_code_links":1,"syntology":null},{"paper":null,"title":"Improving ENIGMA-Style Clause Selection While Learning From History","date":"2021-02-26","arxiv_id":"2102.13564","n_code_links":0,"syntology":null},{"paper":"/paper/towards-automatic-evaluation-of-dialog","title":"Towards Automatic Evaluation of Dialog Systems: A Model-Free Off-Policy Evaluation Approach","date":"2021-02-20","arxiv_id":"2102.10242","n_code_links":1,"syntology":null}],"papers_shown":7,"tasks":[{"task":"/task/automated-theorem-proving","name":"Automated Theorem Proving","papers":2},{"task":"/task/classification-1","name":"Classification","papers":1},{"task":"/task/diagnostic","name":"Diagnostic","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/graph-neural-network","name":"Graph Neural Network","papers":1},{"task":"/task/model-based-reinforcement-learning","name":"Model-based Reinforcement Learning","papers":1},{"task":"/task/off-policy-evaluation","name":"Off-policy evaluation","papers":1},{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":1},{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":1},{"task":"/task/text-generation","name":"Text Generation","papers":1},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":1}],"tasks_shown":11,"n_tasks":11,"usage_by_year":[{"year":"2021","papers":4},{"year":"2022","papers":2},{"year":"2023","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/enigma"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}