{"url":"/task/nlg-evaluation","name":"nlg evaluation","slug":"nlg-evaluation","description_markdown":"Evaluate the generated text by NLG (Natural Language Generation) systems, like large language models","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":71,"papers_with_code":31,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":31,"tagged_in_all":71,"items":[{"url":"/paper/nlg-evaluation-metrics-beyond-correlation","title":"NLG Evaluation Metrics Beyond Correlation Analysis: An Empirical Metric Preference Checklist","date":"2023-05-15","arxiv_id":"2305.08566","repositories_listed":7,"syntology":null},{"url":"/paper/gpteval-nlg-evaluation-using-gpt-4-with","title":"G-Eval: NLG Evaluation using GPT-4 with Better Human Alignment","date":"2023-03-29","arxiv_id":"2303.16634","repositories_listed":3,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/are-llm-based-evaluators-confusing-nlg","title":"Are LLM-based Evaluators Confusing NLG Quality Criteria?","date":"2024-02-19","arxiv_id":"2402.12055","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/not-all-metrics-are-guilty-improving-nlg","title":"Not All Metrics Are Guilty: Improving NLG Evaluation by Diversifying References","date":"2023-05-24","arxiv_id":"2305.15067","repositories_listed":2,"syntology":null},{"url":"/paper/towards-a-unified-multi-dimensional-evaluator","title":"Towards a Unified Multi-Dimensional Evaluator for Text Generation","date":"2022-10-13","arxiv_id":"2210.07197","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/openlgauge-an-explainable-metric-for-nlg","title":"OpeNLGauge: An Explainable Metric for NLG Evaluation with Open-Weights LLMs","date":"2025-03-14","arxiv_id":"2503.11858","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-and-evaluating-correlation-measures","title":"Analyzing and Evaluating Correlation Measures in NLG Meta-Evaluation","date":"2024-10-22","arxiv_id":"2410.16834","repositories_listed":1,"syntology":null},{"url":"/paper/themis-towards-flexible-and-interpretable-nlg","title":"Themis: A Reference-free NLG Evaluation Language Model with Flexibility and Interpretability","date":"2024-06-26","arxiv_id":"2406.18365","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/better-than-random-reliable-nlg-human","title":"Better than Random: Reliable NLG Human Evaluation with Constrained Active Sampling","date":"2024-06-12","arxiv_id":"2406.07967","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/defining-and-detecting-vulnerability-in-human","title":"Defining and Detecting Vulnerability in Human Evaluation Guidelines: A Preliminary Study Towards Reliable NLG Evaluation","date":"2024-06-12","arxiv_id":"2406.07935","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-the-achilles-heel-of-nlg-evaluators","title":"Unveiling the Achilles' Heel of NLG Evaluators: A Unified Adversarial Framework Driven by Large Language Models","date":"2024-05-23","arxiv_id":"2405.14646","repositories_listed":1,"syntology":null},{"url":"/paper/debate-devil-s-advocate-based-assessment-and","title":"DEBATE: Devil's Advocate-Based Assessment and Text Evaluation","date":"2024-05-16","arxiv_id":"2405.09935","repositories_listed":1,"syntology":null},{"url":"/paper/one-prompt-to-rule-them-all-llms-for-opinion","title":"One Prompt To Rule Them All: LLMs for Opinion Summary Evaluation","date":"2024-02-18","arxiv_id":"2402.11683","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/leveraging-large-language-models-for-nlg","title":"Leveraging Large Language Models for NLG Evaluation: Advances and Challenges","date":"2024-01-13","arxiv_id":"2401.07103","repositories_listed":1,"syntology":null},{"url":"/paper/luna-a-framework-for-language-understanding","title":"LUNA: A Framework for Language Understanding and Naturalness Assessment","date":"2024-01-09","arxiv_id":"2401.04522","repositories_listed":1,"syntology":null},{"url":"/paper/towards-multiple-references-era-addressing","title":"Towards Multiple References Era -- Addressing Data Leakage and Limited Reference Diversity in NLG Evaluation","date":"2023-08-06","arxiv_id":"2308.03131","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-nlg-evaluation-through-pairware","title":"LLM Comparative Assessment: Zero-shot NLG Evaluation through Pairwise Comparisons using Large Language Models","date":"2023-07-15","arxiv_id":"2307.07889","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/decompeval-evaluating-generated-texts-as","title":"DecompEval: Evaluating Generated Texts as Unsupervised Decomposed Question Answering","date":"2023-07-13","arxiv_id":"2307.06869","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-nlg-evaluation-metrics-a","title":"Evaluating Evaluation Metrics: A Framework for Analyzing NLG Evaluation Metrics using Measurement Theory","date":"2023-05-24","arxiv_id":"2305.14889","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/is-chatgpt-a-good-nlg-evaluator-a-preliminary","title":"Is ChatGPT a Good NLG Evaluator? A Preliminary Study","date":"2023-03-07","arxiv_id":"2303.04048","repositories_listed":1,"syntology":null},{"url":"/paper/describe-me-an-aucklet-generating-grounded","title":"Describe me an Aucklet: Generating Grounded Perceptual Category Descriptions","date":"2023-03-07","arxiv_id":"2303.04053","repositories_listed":1,"syntology":null},{"url":"/paper/clse-corpus-of-linguistically-significant","title":"CLSE: Corpus of Linguistically Significant Entities","date":"2022-11-04","arxiv_id":"2211.02423","repositories_listed":1,"syntology":null},{"url":"/paper/not-all-errors-are-equal-learning-text","title":"Not All Errors are Equal: Learning Text Generation Metrics using Stratified Error Synthesis","date":"2022-10-10","arxiv_id":"2210.05035","repositories_listed":1,"syntology":null},{"url":"/paper/can-we-do-that-simpler-simple-efficient-high","title":"EffEval: A Comprehensive Evaluation of Efficiency for MT Evaluation Metrics","date":"2022-09-20","arxiv_id":"2209.09593","repositories_listed":1,"syntology":null},{"url":"/paper/near-negative-distinction-giving-a-second","title":"Near-Negative Distinction: Giving a Second Life to Human Evaluation Datasets","date":"2022-05-13","arxiv_id":"2205.06871","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-cross-lingual-gaps-during-leveraging","title":"Bridging Cross-Lingual Gaps During Leveraging the Multilingual Sequence-to-Sequence Pretraining for Text Generation and Understanding","date":"2022-04-16","arxiv_id":"2204.07834","repositories_listed":1,"syntology":null},{"url":"/paper/active-evaluation-efficient-nlg-evaluation-1","title":"Active Evaluation: Efficient NLG Evaluation with Few Pairwise Comparisons","date":"2022-03-11","arxiv_id":"2203.06063","repositories_listed":1,"syntology":null},{"url":"/paper/compression-transduction-and-creation-a","title":"Compression, Transduction, and Creation: A Unified Framework for Evaluating Natural Language Generation","date":"2021-09-14","arxiv_id":"2109.06379","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_unverified":5,"n_pointer_only":0}},{"url":"/paper/perturbation-checklists-for-evaluating-nlg","title":"Perturbation CheckLists for Evaluating NLG Evaluation Metrics","date":"2021-09-13","arxiv_id":"2109.05771","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-of-automatic-metrics-for-the","title":"A Study of Automatic Metrics for the Evaluation of Natural Language Explanations","date":"2021-03-15","arxiv_id":"2103.08545","repositories_listed":1,"syntology":null}],"syntology_records":9,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}