{"url":"/task/hallucination-evaluation","name":"Hallucination Evaluation","slug":"hallucination-evaluation","description_markdown":"Evaluate the ability of LLM to generate non-hallucination text or assess the capability of LLM to recognize hallucinations.","categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":49,"papers_with_code":29,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/hallueditbench","name":"HalluEditBench","full_name":"","num_papers_in_archive":2},{"url":"/dataset/phd","name":"PhD","full_name":"PhD: A ChatGPT-Prompted Visual hallucination Evaluation Dataset","num_papers_in_archive":1},{"url":"/dataset/xinhuahallucinations","name":"UHGEvalDataset","full_name":"","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":29,"of":29,"tagged_in_all":49,"items":[{"url":"/paper/hallusionbench-you-see-what-you-think-or-you","title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","date":"2023-10-23","arxiv_id":"2310.14566","repositories_listed":9,"syntology":{"n":8,"n_ran":4,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/autohallusion-automatic-generation-of","title":"AutoHallusion: Automatic Generation of Hallucination Benchmarks for Vision-Language Models","date":"2024-06-16","arxiv_id":"2406.10900","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/helma-a-large-scale-hallucination-evaluation","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","date":"2023-05-19","arxiv_id":"2305.11747","repositories_listed":3,"syntology":{"n":12,"n_ran":3,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/alleviating-hallucinations-of-large-language","title":"Alleviating Hallucinations of Large Language Models through Induced Hallucinations","date":"2023-12-25","arxiv_id":"2312.15710","repositories_listed":2,"syntology":{"n":13,"n_ran":4,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/knowrl-exploring-knowledgeable-reinforcement","title":"KnowRL: Exploring Knowledgeable Reinforcement Learning for Factuality","date":"2025-06-24","arxiv_id":"2506.19807","repositories_listed":1,"syntology":null},{"url":"/paper/multihal-multilingual-dataset-for-knowledge","title":"MultiHal: Multilingual Dataset for Knowledge-Graph Grounded Evaluation of LLM Hallucinations","date":"2025-05-20","arxiv_id":"2505.14101","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-llm-faithfulness-in-rag-with","title":"Benchmarking LLM Faithfulness in RAG with Evolving Leaderboards","date":"2025-05-07","arxiv_id":"2505.04847","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/exploring-hallucination-of-large-multimodal","title":"Exploring Hallucination of Large Multimodal Models in Video Understanding: Benchmark, Analysis and Mitigation","date":"2025-03-25","arxiv_id":"2503.19622","repositories_listed":1,"syntology":null},{"url":"/paper/2503-01670","title":"Evaluating LLMs' Assessment of Mixed-Context Hallucination Through the Lens of Summarization","date":"2025-03-03","arxiv_id":"2503.01670","repositories_listed":1,"syntology":null},{"url":"/paper/treecut-a-synthetic-unanswerable-math-word","title":"TreeCut: A Synthetic Unanswerable Math Word Problem Dataset for LLM Hallucination Evaluation","date":"2025-02-19","arxiv_id":"2502.13442","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/dahl-domain-specific-automated-hallucination","title":"DAHL: Domain-specific Automated Hallucination Evaluation of Long-Form Text through a Benchmark Dataset in Biomedicine","date":"2024-11-14","arxiv_id":"2411.09255","repositories_listed":1,"syntology":null},{"url":"/paper/ddfav-remote-sensing-large-vision-language","title":"DDFAV: Remote Sensing Large Vision Language Models Dataset and Evaluation Benchmark","date":"2024-11-05","arxiv_id":"2411.02733","repositories_listed":1,"syntology":null},{"url":"/paper/unified-triplet-level-hallucination","title":"Unified Triplet-Level Hallucination Evaluation for Large Vision-Language Models","date":"2024-10-30","arxiv_id":"2410.23114","repositories_listed":1,"syntology":null},{"url":"/paper/longhalqa-long-context-hallucination","title":"LongHalQA: Long-Context Hallucination Evaluation for MultiModal Large Language Models","date":"2024-10-13","arxiv_id":"2410.09962","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-image-hallucination-in-text-to","title":"Evaluating Image Hallucination in Text-to-Image Generation with Question-Answering","date":"2024-09-19","arxiv_id":"2409.12784","repositories_listed":1,"syntology":null},{"url":"/paper/reefknot-a-comprehensive-benchmark-for","title":"Reefknot: A Comprehensive Benchmark for Relation Hallucination Evaluation, Analysis and Mitigation in Multimodal Large Language Models","date":"2024-08-18","arxiv_id":"2408.09429","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/enhancing-llm-s-cognition-via-structurization","title":"Enhancing LLM's Cognition via Structurization","date":"2024-07-23","arxiv_id":"2407.16434","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/defan-definitive-answer-dataset-for-llms","title":"DefAn: Definitive Answer Dataset for LLMs Hallucination Evaluation","date":"2024-06-13","arxiv_id":"2406.09155","repositories_listed":1,"syntology":null},{"url":"/paper/phd-a-prompted-visual-hallucination","title":"PhD: A ChatGPT-Prompted Visual hallucination Evaluation Dataset","date":"2024-03-17","arxiv_id":"2403.11116","repositories_listed":1,"syntology":null},{"url":"/paper/diahalu-a-dialogue-level-hallucination","title":"DiaHalu: A Dialogue-level Hallucination Evaluation Benchmark for Large Language Models","date":"2024-03-01","arxiv_id":"2403.00896","repositories_listed":1,"syntology":null},{"url":"/paper/truthx-alleviating-hallucinations-by-editing","title":"TruthX: Alleviating Hallucinations by Editing Large Language Models in Truthful Space","date":"2024-02-27","arxiv_id":"2402.17811","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/mitigating-fine-grained-hallucination-by-fine","title":"Mitigating Fine-Grained Hallucination by Fine-Tuning Large Vision-Language Models with Caption Rewrites","date":"2023-12-04","arxiv_id":"2312.01701","repositories_listed":1,"syntology":null},{"url":"/paper/uhgeval-benchmarking-the-hallucination-of","title":"UHGEval: Benchmarking the Hallucination of Chinese Large Language Models via Unconstrained Generation","date":"2023-11-26","arxiv_id":"2311.15296","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/hallucidoctor-mitigating-hallucinatory","title":"HalluciDoctor: Mitigating Hallucinatory Toxicity in Visual Instruction Data","date":"2023-11-22","arxiv_id":"2311.13614","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_unverified":6,"n_pointer_only":7}},{"url":"/paper/lighter-yet-more-faithful-investigating","title":"Investigating Hallucinations in Pruned Large Language Models for Abstractive Summarization","date":"2023-11-15","arxiv_id":"2311.09335","repositories_listed":1,"syntology":null},{"url":"/paper/an-llm-free-multi-dimensional-benchmark-for","title":"AMBER: An LLM-free Multi-dimensional Benchmark for MLLMs Hallucination Evaluation","date":"2023-11-13","arxiv_id":"2311.07397","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/analyzing-and-mitigating-object-hallucination","title":"Analyzing and Mitigating Object Hallucination in Large Vision-Language Models","date":"2023-10-01","arxiv_id":"2310.00754","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_unverified":1,"n_pointer_only":8}},{"url":"/paper/evaluation-and-analysis-of-hallucination-in","title":"Evaluation and Analysis of Hallucination in Large Vision-Language Models","date":"2023-08-29","arxiv_id":"2308.15126","repositories_listed":1,"syntology":null},{"url":"/paper/mindmap-knowledge-graph-prompting-sparks","title":"MindMap: Knowledge Graph Prompting Sparks Graph of Thoughts in Large Language Models","date":"2023-08-17","arxiv_id":"2308.09729","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_unverified":2,"n_pointer_only":8}}],"syntology_records":14,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}