{"url":"/task/hallucination","name":"Hallucination","slug":"hallucination","description_markdown":null,"categories":[],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1816,"papers_with_code":752,"benchmarks":0,"benchmark_tables_in_archive":0,"benchmark_tables_shown":0,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":0},"benchmarks":[],"datasets":[{"url":"/dataset/mmneedle","name":"MMNeedle","full_name":"Multimodal Needle in a Haystack","num_papers_in_archive":12}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":752,"tagged_in_all":1816,"items":[{"url":"/paper/pulse-self-supervised-photo-upsampling-via","title":"PULSE: Self-Supervised Photo Upsampling via Latent Space Exploration of Generative Models","date":"2020-03-08","arxiv_id":"2003.03808","repositories_listed":16,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/hallusionbench-you-see-what-you-think-or-you","title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","date":"2023-10-23","arxiv_id":"2310.14566","repositories_listed":9,"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":1}},{"url":"/paper/react-synergizing-reasoning-and-acting-in","title":"ReAct: Synergizing Reasoning and Acting in Language Models","date":"2022-10-06","arxiv_id":"2210.03629","repositories_listed":9,"syntology":{"n":34,"n_ran":15,"n_unverified":19,"n_pointer_only":5}},{"url":"/paper/mitigating-object-hallucinations-in-large","title":"Mitigating Object Hallucinations in Large Vision-Language Models through Visual Contrastive Decoding","date":"2023-11-28","arxiv_id":"2311.16922","repositories_listed":7,"syntology":{"n":14,"n_ran":9,"n_unverified":5,"n_pointer_only":1}},{"url":"/paper/evaluating-object-hallucination-in-large","title":"Evaluating Object Hallucination in Large Vision-Language Models","date":"2023-05-17","arxiv_id":"2305.10355","repositories_listed":6,"syntology":{"n":12,"n_ran":6,"n_unverified":6,"n_pointer_only":4}},{"url":"/paper/rlaif-v-aligning-mllms-through-open-source-ai","title":"RLAIF-V: Open-Source AI Feedback Leads to Super GPT-4V Trustworthiness","date":"2024-05-27","arxiv_id":"2405.17220","repositories_listed":5,"syntology":{"n":22,"n_ran":14,"n_unverified":8,"n_pointer_only":8}},{"url":"/paper/retrieval-augmented-generation-for-large","title":"Retrieval-Augmented Generation for Large Language Models: A Survey","date":"2023-12-18","arxiv_id":"2312.10997","repositories_listed":4,"syntology":null},{"url":"/paper/rlhf-v-towards-trustworthy-mllms-via-behavior","title":"RLHF-V: Towards Trustworthy MLLMs via Behavior Alignment from Fine-grained Correctional Human Feedback","date":"2023-12-01","arxiv_id":"2312.00849","repositories_listed":4,"syntology":null},{"url":"/paper/aligning-large-multi-modal-model-with-robust","title":"Mitigating Hallucination in Large Multi-Modal Models via Robust Instruction Tuning","date":"2023-06-26","arxiv_id":"2306.14565","repositories_listed":4,"syntology":null},{"url":"/paper/projected-distribution-loss-for-image","title":"Projected Distribution Loss for Image Enhancement","date":"2020-12-16","arxiv_id":"2012.09289","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/pushing-the-limits-of-low-resource","title":"Pushing the Limits of Low-Resource Morphological Inflection","date":"2019-08-16","arxiv_id":"1908.05838","repositories_listed":4,"syntology":null},{"url":"/paper/im2flow-motion-hallucination-from-static","title":"Im2Flow: Motion Hallucination from Static Images for Action Recognition","date":"2017-12-12","arxiv_id":"1712.04109","repositories_listed":4,"syntology":null},{"url":"/paper/evolutionary-thoughts-integration-of-large","title":"Evolutionary thoughts: integration of large language models and evolutionary algorithms","date":"2025-05-09","arxiv_id":"2505.05756","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/benchmark-evaluations-applications-and","title":"A Survey of State of the Art Large Vision Language Models: Alignment, Benchmark, Evaluations and Challenges","date":"2025-01-04","arxiv_id":"2501.02189","repositories_listed":3,"syntology":null},{"url":"/paper/embodied-agent-interface-benchmarking-llms","title":"Embodied Agent Interface: Benchmarking LLMs for Embodied Decision Making","date":"2024-10-09","arxiv_id":"2410.07166","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/swift-a-scalable-lightweight-infrastructure","title":"SWIFT:A Scalable lightWeight Infrastructure for Fine-Tuning","date":"2024-08-10","arxiv_id":"2408.05517","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/autohallusion-automatic-generation-of","title":"AutoHallusion: Automatic Generation of Hallucination Benchmarks for Vision-Language Models","date":"2024-06-16","arxiv_id":"2406.10900","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/prismatic-vlms-investigating-the-design-space","title":"Prismatic VLMs: Investigating the Design Space of Visually-Conditioned Language Models","date":"2024-02-12","arxiv_id":"2402.07865","repositories_listed":3,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":4}},{"url":"/paper/moe-llava-mixture-of-experts-for-large-vision","title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","date":"2024-01-29","arxiv_id":"2401.15947","repositories_listed":3,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/ragtruth-a-hallucination-corpus-for","title":"RAGTruth: A Hallucination Corpus for Developing Trustworthy Retrieval-Augmented Language Models","date":"2023-12-31","arxiv_id":"2401.00396","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/evaluating-hallucinations-in-chinese-large","title":"Evaluating Hallucinations in Chinese Large Language Models","date":"2023-10-05","arxiv_id":"2310.03368","repositories_listed":3,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/think-on-graph-deep-and-responsible-reasoning","title":"Think-on-Graph: Deep and Responsible Reasoning of Large Language Model on Knowledge Graph","date":"2023-07-15","arxiv_id":"2307.07697","repositories_listed":3,"syntology":{"n":16,"n_ran":11,"n_unverified":5,"n_pointer_only":15}},{"url":"/paper/helma-a-large-scale-hallucination-evaluation","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","date":"2023-05-19","arxiv_id":"2305.11747","repositories_listed":3,"syntology":{"n":12,"n_ran":3,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/multimodal-chain-of-thought-reasoning-in","title":"Multimodal Chain-of-Thought Reasoning in Language Models","date":"2023-02-02","arxiv_id":"2302.00923","repositories_listed":3,"syntology":{"n":11,"n_ran":6,"n_unverified":5,"n_pointer_only":1}},{"url":"/paper/dataset-distillation-via-factorization","title":"Dataset Distillation via Factorization","date":"2022-10-30","arxiv_id":"2210.16774","repositories_listed":3,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/on-hallucinations-in-tomographic-image","title":"On hallucinations in tomographic image reconstruction","date":"2020-12-01","arxiv_id":"2012.00646","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/discosg-towards-discourse-level-text-scene","title":"DiscoSG: Towards Discourse-Level Text Scene Graph Parsing through Iterative Graph Refinement","date":"2025-06-18","arxiv_id":"2506.15583","repositories_listed":2,"syntology":{"n":19,"n_ran":1,"n_unverified":18,"n_pointer_only":19}},{"url":"/paper/lettucedetect-a-hallucination-detection","title":"LettuceDetect: A Hallucination Detection Framework for RAG Applications","date":"2025-02-24","arxiv_id":"2502.17125","repositories_listed":2,"syntology":null},{"url":"/paper/fixing-imbalanced-attention-to-mitigate-in","title":"PAINT: Paying Attention to INformed Tokens to Mitigate Hallucination in Large Vision-Language Model","date":"2025-01-21","arxiv_id":"2501.12206","repositories_listed":2,"syntology":null},{"url":"/paper/faithbench-a-diverse-hallucination-benchmark","title":"FaithBench: A Diverse Hallucination Benchmark for Summarization by Modern LLMs","date":"2024-10-17","arxiv_id":"2410.13210","repositories_listed":2,"syntology":null}],"syntology_records":21,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}