{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/hallucination/papers/3","list_of":"/task/hallucination","task":"Hallucination","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":19,"rows_per_page":100,"rows":[201,300],"of":1816,"counts":{"archive_papers_tagged":1816,"with_a_code_link":752,"where_syntology_ran_a_sample":276,"not_listed_spam_title":0,"listed":1816,"listed_where_code_ran":276,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":240,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":240,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/hallucination","prev":"/task/hallucination/papers/2","next":"/task/hallucination/papers/4","papers":[{"url":"/paper/knowledge-graph-guided-retrieval-augmented","slug":"knowledge-graph-guided-retrieval-augmented","title":"Knowledge Graph-Guided Retrieval Augmented Generation","date":"2025-02-08","arxiv_id":"2502.06864","repositories_listed":1,"syntology":null},{"url":"/paper/learning-conformal-abstention-policies-for","slug":"learning-conformal-abstention-policies-for","title":"Learning Conformal Abstention Policies for Adaptive Risk Management in Large Language and Vision-Language Models","date":"2025-02-08","arxiv_id":"2502.06884","repositories_listed":1,"syntology":{"n":16,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/learning-conformal-abstention-policies-for#ran","syntology_url":"https://syntology.ai/paper/2502.06884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06884"}},"official":{"repos":["sinatayebati/vlm-uncertainty"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/self-rationalization-in-the-wild-a-large","slug":"self-rationalization-in-the-wild-a-large","title":"Self-Rationalization in the Wild: A Large Scale Out-of-Distribution Evaluation on NLI-related tasks","date":"2025-02-07","arxiv_id":"2502.04797","repositories_listed":1,"syntology":null},{"url":"/paper/videorope-what-makes-for-good-video-rotary","slug":"videorope-what-makes-for-good-video-rotary","title":"VideoRoPE: What Makes for Good Video Rotary Position Embedding?","date":"2025-02-07","arxiv_id":"2502.05173","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videorope-what-makes-for-good-video-rotary#ran","syntology_url":"https://syntology.ai/paper/2502.05173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05173"}},"official":{"repos":["wiselnn570/videorope"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-for-multi-robot-systems","slug":"large-language-models-for-multi-robot-systems","title":"Large Language Models for Multi-Robot Systems: A Survey","date":"2025-02-06","arxiv_id":"2502.03814","repositories_listed":1,"syntology":null},{"url":"/paper/linear-correlation-in-lm-s-compositional","slug":"linear-correlation-in-lm-s-compositional","title":"Linear Correlation in LM's Compositional Generalization and Hallucination","date":"2025-02-06","arxiv_id":"2502.04520","repositories_listed":1,"syntology":null},{"url":"/paper/the-hidden-life-of-tokens-reducing","slug":"the-hidden-life-of-tokens-reducing","title":"The Hidden Life of Tokens: Reducing Hallucination of Large Vision-Language Models via Visual Information Steering","date":"2025-02-05","arxiv_id":"2502.03628","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-hidden-life-of-tokens-reducing#ran","syntology_url":"https://syntology.ai/paper/2502.03628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.03628"}},"official":{"repos":["LzVv123456/VISTA"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/damo-data-and-model-aware-alignment-of-multi","slug":"damo-data-and-model-aware-alignment-of-multi","title":"DAMO: Data- and Model-aware Alignment of Multi-modal LLMs","date":"2025-02-04","arxiv_id":"2502.01943","repositories_listed":1,"syntology":null},{"url":"/paper/differentially-private-steering-for-large","slug":"differentially-private-steering-for-large","title":"Differentially Private Steering for Large Language Model Alignment","date":"2025-01-30","arxiv_id":"2501.18532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/differentially-private-steering-for-large#ran","syntology_url":"https://syntology.ai/paper/2501.18532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.18532"}},"official":{"repos":["ukplab/iclr2025-psa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chip-cross-modal-hierarchical-direct","slug":"chip-cross-modal-hierarchical-direct","title":"CHiP: Cross-modal Hierarchical Direct Preference Optimization for Multimodal LLMs","date":"2025-01-28","arxiv_id":"2501.16629","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/chip-cross-modal-hierarchical-direct#ran","syntology_url":"https://syntology.ai/paper/2501.16629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.16629"}},"official":{"repos":["lvugai/chip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-think-on-graph-wider-deeper-and-faster","slug":"fast-think-on-graph-wider-deeper-and-faster","title":"Fast Think-on-Graph: Wider, Deeper and Faster Reasoning of Large Language Model on Knowledge Graph","date":"2025-01-24","arxiv_id":"2501.14300","repositories_listed":1,"syntology":null},{"url":"/paper/onioneval-an-unified-evaluation-of-fact","slug":"onioneval-an-unified-evaluation-of-fact","title":"OnionEval: An Unified Evaluation of Fact-conflicting Hallucination for Small-Large Language Models","date":"2025-01-22","arxiv_id":"2501.12975","repositories_listed":1,"syntology":null},{"url":"/paper/hallucination-mitigation-using-agentic-ai","slug":"hallucination-mitigation-using-agentic-ai","title":"Hallucination Mitigation using Agentic AI Natural Language-Based Frameworks","date":"2025-01-19","arxiv_id":"2501.13946","repositories_listed":1,"syntology":null},{"url":"/paper/chartinsighter-an-approach-for-mitigating","slug":"chartinsighter-an-approach-for-mitigating","title":"ChartInsighter: An Approach for Mitigating Hallucination in Time-series Chart Summary Generation with A Benchmark Dataset","date":"2025-01-16","arxiv_id":"2501.09349","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-hallucinations-in-large-vision-3","slug":"mitigating-hallucinations-in-large-vision-3","title":"Mitigating Hallucinations in Large Vision-Language Models via DPO: On-Policy Data Hold the Key","date":"2025-01-16","arxiv_id":"2501.09695","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mitigating-hallucinations-in-large-vision-3#ran","syntology_url":"https://syntology.ai/paper/2501.09695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.09695"}},"official":{"repos":["zhyang2226/opa-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-graph-based-retrieval-augmented","slug":"knowledge-graph-based-retrieval-augmented","title":"Knowledge Graph-based Retrieval-Augmented Generation for Schema Matching","date":"2025-01-15","arxiv_id":"2501.08686","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-llms-can-reason-about-aesthetics","slug":"multimodal-llms-can-reason-about-aesthetics","title":"Multimodal LLMs Can Reason about Aesthetics in Zero-Shot","date":"2025-01-15","arxiv_id":"2501.09012","repositories_listed":1,"syntology":null},{"url":"/paper/tarsier2-advancing-large-vision-language","slug":"tarsier2-advancing-large-vision-language","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","date":"2025-01-14","arxiv_id":"2501.07888","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier2-advancing-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2501.07888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.07888"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/fine-tuning-large-language-models-for-6","slug":"fine-tuning-large-language-models-for-6","title":"Fine-tuning Large Language Models for Improving Factuality in Legal Question Answering","date":"2025-01-11","arxiv_id":"2501.06521","repositories_listed":1,"syntology":null},{"url":"/paper/vasparse-towards-efficient-visual","slug":"vasparse-towards-efficient-visual","title":"VASparse: Towards Efficient Visual Hallucination Mitigation for Large Vision-Language Model via Visual-Aware Sparsification","date":"2025-01-11","arxiv_id":"2501.06553","repositories_listed":1,"syntology":null},{"url":"/paper/ecbench-can-multi-modal-foundation-models","slug":"ecbench-can-multi-modal-foundation-models","title":"ECBench: Can Multi-modal Foundation Models Understand the Egocentric World? A Holistic Embodied Cognition Benchmark","date":"2025-01-09","arxiv_id":"2501.05031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ecbench-can-multi-modal-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2501.05031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05031"}},"official":{"repos":["rh-dang/ecbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/socratic-questioning-learn-to-self-guide","slug":"socratic-questioning-learn-to-self-guide","title":"Socratic Questioning: Learn to Self-guide Multimodal Reasoning in the Wild","date":"2025-01-06","arxiv_id":"2501.02964","repositories_listed":1,"syntology":null},{"url":"/paper/chair-classifier-of-hallucination-as-improver","slug":"chair-classifier-of-hallucination-as-improver","title":"CHAIR -- Classifier of Hallucination as Improver","date":"2025-01-05","arxiv_id":"2501.02518","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-hallucination-for-large-vision","slug":"mitigating-hallucination-for-large-vision","title":"Mitigating Hallucination for Large Vision Language Model by Inter-Modality Correlation Calibration Decoding","date":"2025-01-03","arxiv_id":"2501.01926","repositories_listed":1,"syntology":null},{"url":"/paper/think-more-hallucinate-less-mitigating","slug":"think-more-hallucinate-less-mitigating","title":"Think More, Hallucinate Less: Mitigating Hallucinations via Dual Process of Fast and Slow Thinking","date":"2025-01-02","arxiv_id":"2501.01306","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-object-hallucinations-in-large-2","slug":"mitigating-object-hallucinations-in-large-2","title":"Mitigating Object Hallucinations in Large Vision-Language Models with Assembly of Global and Local Attention","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/octopus-alleviating-hallucination-via-dynamic","slug":"octopus-alleviating-hallucination-via-dynamic","title":"Octopus: Alleviating Hallucination via Dynamic Contrastive Decoding","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rrhf-v-ranking-responses-to-mitigate","slug":"rrhf-v-ranking-responses-to-mitigate","title":"RRHF-V: Ranking Responses to Mitigate Hallucinations in Multimodal Large Language Models with Human Feedback","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/vasparse-towards-efficient-visual-1","slug":"vasparse-towards-efficient-visual-1","title":"VASparse: Towards Efficient Visual Hallucination Mitigation via Visual-Aware Token Sparsification","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/hallucinogen-a-benchmark-for-evaluating","slug":"hallucinogen-a-benchmark-for-evaluating","title":"HALLUCINOGEN: A Benchmark for Evaluating Object Hallucination in Large Visual-Language Models","date":"2024-12-29","arxiv_id":"2412.20622","repositories_listed":1,"syntology":null},{"url":"/paper/extract-free-dense-misalignment-from-clip","slug":"extract-free-dense-misalignment-from-clip","title":"Extract Free Dense Misalignment from CLIP","date":"2024-12-24","arxiv_id":"2412.18404","repositories_listed":1,"syntology":null},{"url":"/paper/harnessing-large-language-models-for-1","slug":"harnessing-large-language-models-for-1","title":"Harnessing Large Language Models for Knowledge Graph Question Answering via Adaptive Multi-Aspect Retrieval-Augmentation","date":"2024-12-24","arxiv_id":"2412.18537","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-large-language-models-for-1#ran","syntology_url":"https://syntology.ai/paper/2412.18537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18537"}},"official":{"repos":["Applied-Machine-Learning-Lab/AMAR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/citebart-learning-to-generate-citations-for","slug":"citebart-learning-to-generate-citations-for","title":"CiteBART: Learning to Generate Citations for Local Citation Recommendation","date":"2024-12-23","arxiv_id":"2412.17534","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/citebart-learning-to-generate-citations-for#ran","syntology_url":"https://syntology.ai/paper/2412.17534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.17534"}},"official":{"repos":["eyclk/citationrecommendation"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-preference-data-synthetic","slug":"multimodal-preference-data-synthetic","title":"Multimodal Preference Data Synthetic Alignment with Reward Model","date":"2024-12-23","arxiv_id":"2412.17417","repositories_listed":1,"syntology":null},{"url":"/paper/a-benchmark-and-robustness-study-of-in","slug":"a-benchmark-and-robustness-study-of-in","title":"A Benchmark and Robustness Study of In-Context-Learning with Large Language Models in Music Entity Detection","date":"2024-12-16","arxiv_id":"2412.11851","repositories_listed":1,"syntology":null},{"url":"/paper/emma-x-an-embodied-multimodal-action-model","slug":"emma-x-an-embodied-multimodal-action-model","title":"Emma-X: An Embodied Multimodal Action Model with Grounded Chain of Thought and Look-ahead Spatial Reasoning","date":"2024-12-16","arxiv_id":"2412.11974","repositories_listed":1,"syntology":null},{"url":"/paper/filter-then-generate-large-language-models","slug":"filter-then-generate-large-language-models","title":"Filter-then-Generate: Large Language Models with Structure-Text Adapter for Knowledge Graph Completion","date":"2024-12-12","arxiv_id":"2412.09094","repositories_listed":1,"syntology":null},{"url":"/paper/granite-guardian","slug":"granite-guardian","title":"Granite Guardian","date":"2024-12-10","arxiv_id":"2412.07724","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/granite-guardian#ran","syntology_url":"https://syntology.ai/paper/2412.07724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07724"}},"official":{"repos":["ibm-granite/granite-guardian"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hallucination-elimination-and-semantic","slug":"hallucination-elimination-and-semantic","title":"Hallucination Elimination and Semantic Enhancement Framework for Vision-Language Models in Traffic Scenarios","date":"2024-12-10","arxiv_id":"2412.07518","repositories_listed":1,"syntology":null},{"url":"/paper/delve-into-visual-contrastive-decoding-for","slug":"delve-into-visual-contrastive-decoding-for","title":"Delve into Visual Contrastive Decoding for Hallucination Mitigation of Large Vision-Language Models","date":"2024-12-09","arxiv_id":"2412.06775","repositories_listed":1,"syntology":null},{"url":"/paper/expanding-performance-boundaries-of-open","slug":"expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","arxiv_id":"2412.05271","repositories_listed":1,"syntology":{"n":9,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/expanding-performance-boundaries-of-open#ran","syntology_url":"https://syntology.ai/paper/2412.05271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05271"}},"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/florence-vl-enhancing-vision-language-models","slug":"florence-vl-enhancing-vision-language-models","title":"Florence-VL: Enhancing Vision-Language Models with Generative Vision Encoder and Depth-Breadth Fusion","date":"2024-12-05","arxiv_id":"2412.04424","repositories_listed":1,"syntology":null},{"url":"/paper/automating-feedback-analysis-in-surgical","slug":"automating-feedback-analysis-in-surgical","title":"Automating Feedback Analysis in Surgical Training: Detection, Categorization, and Assessment","date":"2024-12-01","arxiv_id":"2412.00760","repositories_listed":1,"syntology":null},{"url":"/paper/can-llms-be-good-graph-judger-for-knowledge","slug":"can-llms-be-good-graph-judger-for-knowledge","title":"Can LLMs be Good Graph Judge for Knowledge Graph Construction?","date":"2024-11-26","arxiv_id":"2411.17388","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-llms-be-good-graph-judger-for-knowledge#ran","syntology_url":"https://syntology.ai/paper/2411.17388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17388"}},"official":{"repos":["hhy-huang/graphjudge"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/atomr-atomic-operator-empowered-large","slug":"atomr-atomic-operator-empowered-large","title":"AtomR: Atomic Operator-Empowered Large Language Models for Heterogeneous Knowledge Reasoning","date":"2024-11-25","arxiv_id":"2411.16495","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":13,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/atomr-atomic-operator-empowered-large#ran","syntology_url":"https://syntology.ai/paper/2411.16495","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16495"}},"official":{"repos":["THU-KEG/AtomR"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/o1-replication-journey-part-2-surpassing-o1","slug":"o1-replication-journey-part-2-surpassing-o1","title":"O1 Replication Journey -- Part 2: Surpassing O1-preview through Simple Distillation, Big Progress or Bitter Lesson?","date":"2024-11-25","arxiv_id":"2411.16489","repositories_listed":1,"syntology":null},{"url":"/paper/vidhal-benchmarking-temporal-hallucinations","slug":"vidhal-benchmarking-temporal-hallucinations","title":"VidHal: Benchmarking Temporal Hallucinations in Vision LLMs","date":"2024-11-25","arxiv_id":"2411.16771","repositories_listed":1,"syntology":null},{"url":"/paper/valid-mitigating-the-hallucination-of-large","slug":"valid-mitigating-the-hallucination-of-large","title":"VaLiD: Mitigating the Hallucination of Large Vision Language Models by Visual Layer Fusion Contrastive Decoding","date":"2024-11-24","arxiv_id":"2411.15839","repositories_listed":1,"syntology":null},{"url":"/paper/devils-in-middle-layers-of-large-vision","slug":"devils-in-middle-layers-of-large-vision","title":"Devils in Middle Layers of Large Vision-Language Models: Interpreting, Detecting and Mitigating Object Hallucinations via Attention Lens","date":"2024-11-23","arxiv_id":"2411.16724","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/devils-in-middle-layers-of-large-vision#ran","syntology_url":"https://syntology.ai/paper/2411.16724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16724"}},"official":{"repos":["zhangqijiang07/middle_layers_indicating_hallucinations"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ontology-constrained-generation-of-domain","slug":"ontology-constrained-generation-of-domain","title":"Ontology-Constrained Generation of Domain-Specific Clinical Summaries","date":"2024-11-23","arxiv_id":"2411.15666","repositories_listed":1,"syntology":null},{"url":"/paper/vl-uncertainty-detecting-hallucination-in","slug":"vl-uncertainty-detecting-hallucination-in","title":"VL-Uncertainty: Detecting Hallucination in Large Vision-Language Model via Uncertainty Estimation","date":"2024-11-18","arxiv_id":"2411.11919","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":3,"n_honours":3,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 3 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vl-uncertainty-detecting-hallucination-in#ran","syntology_url":"https://syntology.ai/paper/2411.11919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11919"}},"official":null}},{"url":"/paper/thinking-before-looking-improving-multimodal","slug":"thinking-before-looking-improving-multimodal","title":"Thinking Before Looking: Improving Multimodal LLM Reasoning via Mitigating Visual Hallucination","date":"2024-11-15","arxiv_id":"2411.12591","repositories_listed":1,"syntology":null},{"url":"/paper/dahl-domain-specific-automated-hallucination","slug":"dahl-domain-specific-automated-hallucination","title":"DAHL: Domain-specific Automated Hallucination Evaluation of Long-Form Text through a Benchmark Dataset in Biomedicine","date":"2024-11-14","arxiv_id":"2411.09255","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-the-visual-gap-fine-tuning","slug":"bridging-the-visual-gap-fine-tuning","title":"Bridging the Visual Gap: Fine-Tuning Multimodal Models with Knowledge-Adapted Captions","date":"2024-11-13","arxiv_id":"2411.09018","repositories_listed":1,"syntology":null},{"url":"/paper/confidence-aware-denoised-fine-tuning-of-off","slug":"confidence-aware-denoised-fine-tuning-of-off","title":"Confidence-aware Denoised Fine-tuning of Off-the-shelf Models for Certified Robustness","date":"2024-11-13","arxiv_id":"2411.08933","repositories_listed":1,"syntology":null},{"url":"/paper/decoprompt-decoding-prompts-reduces","slug":"decoprompt-decoding-prompts-reduces","title":"DecoPrompt : Decoding Prompts Reduces Hallucinations when Large Language Models Meet False Premises","date":"2024-11-12","arxiv_id":"2411.07457","repositories_listed":1,"syntology":null},{"url":"/paper/verbosity-neq-veracity-demystify-verbosity","slug":"verbosity-neq-veracity-demystify-verbosity","title":"Verbosity $\\neq$ Veracity: Demystify Verbosity Compensation Behavior of Large Language Models","date":"2024-11-12","arxiv_id":"2411.07858","repositories_listed":1,"syntology":null},{"url":"/paper/assistrag-boosting-the-potential-of-large","slug":"assistrag-boosting-the-potential-of-large","title":"AssistRAG: Boosting the Potential of Large Language Models with an Intelligent Information Assistant","date":"2024-11-11","arxiv_id":"2411.06805","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":1,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"10 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/assistrag-boosting-the-potential-of-large#ran","syntology_url":"https://syntology.ai/paper/2411.06805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.06805"}},"official":{"repos":["smallporridge/assistrag"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":1,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-multimodal-retrieval-augmented","slug":"benchmarking-multimodal-retrieval-augmented","title":"Benchmarking Multimodal Retrieval Augmented Generation with Dynamic VQA Dataset and Self-adaptive Planning Agent","date":"2024-11-05","arxiv_id":"2411.02937","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-multimodal-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2411.02937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02937"}},"official":{"repos":["alibaba-nlp/omnisearch"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ddfav-remote-sensing-large-vision-language","slug":"ddfav-remote-sensing-large-vision-language","title":"DDFAV: Remote Sensing Large Vision Language Models Dataset and Evaluation Benchmark","date":"2024-11-05","arxiv_id":"2411.02733","repositories_listed":1,"syntology":null},{"url":"/paper/htmlrag-html-is-better-than-plain-text-for","slug":"htmlrag-html-is-better-than-plain-text-for","title":"HtmlRAG: HTML is Better Than Plain Text for Modeling Retrieved Knowledge in RAG Systems","date":"2024-11-05","arxiv_id":"2411.02959","repositories_listed":1,"syntology":null},{"url":"/paper/v-dpo-mitigating-hallucination-in-large","slug":"v-dpo-mitigating-hallucination-in-large","title":"V-DPO: Mitigating Hallucination in Large Vision Language Models via Vision-Guided Direct Preference Optimization","date":"2024-11-05","arxiv_id":"2411.02712","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/v-dpo-mitigating-hallucination-in-large#ran","syntology_url":"https://syntology.ai/paper/2411.02712","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02712"}},"official":{"repos":["yuxixie/v-dpo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rate-explain-and-cite-rec-enhanced","slug":"rate-explain-and-cite-rec-enhanced","title":"Rate, Explain and Cite (REC): Enhanced Explanation and Attribution in Automatic Evaluation by Large Language Models","date":"2024-11-03","arxiv_id":"2411.02448","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-ontology-in-dialogue-state-tracking","slug":"beyond-ontology-in-dialogue-state-tracking","title":"Beyond Ontology in Dialogue State Tracking for Goal-Oriented Chatbot","date":"2024-10-30","arxiv_id":"2410.22767","repositories_listed":1,"syntology":null},{"url":"/paper/unified-triplet-level-hallucination","slug":"unified-triplet-level-hallucination","title":"Unified Triplet-Level Hallucination Evaluation for Large Vision-Language Models","date":"2024-10-30","arxiv_id":"2410.23114","repositories_listed":1,"syntology":null},{"url":"/paper/distinguishing-ignorance-from-error-in-llm","slug":"distinguishing-ignorance-from-error-in-llm","title":"Distinguishing Ignorance from Error in LLM Hallucinations","date":"2024-10-29","arxiv_id":"2410.22071","repositories_listed":1,"syntology":null},{"url":"/paper/timesuite-improving-mllms-for-long-video","slug":"timesuite-improving-mllms-for-long-video","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","date":"2024-10-25","arxiv_id":"2410.19702","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/timesuite-improving-mllms-for-long-video#ran","syntology_url":"https://syntology.ai/paper/2410.19702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19702"}},"official":null}},{"url":"/paper/navigating-noisy-feedback-enhancing","slug":"navigating-noisy-feedback-enhancing","title":"Navigating Noisy Feedback: Enhancing Reinforcement Learning with Error-Prone Language Models","date":"2024-10-22","arxiv_id":"2410.17389","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/navigating-noisy-feedback-enhancing#ran","syntology_url":"https://syntology.ai/paper/2410.17389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17389"}},"official":{"repos":["sy-shi/RLAIF_ScoreDiff"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/proverag-provenance-driven-vulnerability","slug":"proverag-provenance-driven-vulnerability","title":"ProveRAG: Provenance-Driven Vulnerability Analysis with Automated Retrieval-Augmented LLMs","date":"2024-10-22","arxiv_id":"2410.17406","repositories_listed":1,"syntology":null},{"url":"/paper/can-knowledge-editing-really-correct","slug":"can-knowledge-editing-really-correct","title":"Can Knowledge Editing Really Correct Hallucinations?","date":"2024-10-21","arxiv_id":"2410.16251","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":14,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/can-knowledge-editing-really-correct#ran","syntology_url":"https://syntology.ai/paper/2410.16251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16251"}},"official":{"repos":["llm-editing/HalluEditBench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-object-hallucination-via","slug":"mitigating-object-hallucination-via","title":"Mitigating Object Hallucination via Concentric Causal Attention","date":"2024-10-21","arxiv_id":"2410.15926","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/mitigating-object-hallucination-via#ran","syntology_url":"https://syntology.ai/paper/2410.15926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15926"}},"official":{"repos":["xing0047/cca-llava"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/reducing-hallucinations-in-vision-language","slug":"reducing-hallucinations-in-vision-language","title":"Reducing Hallucinations in Vision-Language Models via Latent Space Steering","date":"2024-10-21","arxiv_id":"2410.15778","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reducing-hallucinations-in-vision-language#ran","syntology_url":"https://syntology.ai/paper/2410.15778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15778"}},"official":{"repos":["shengliu66/vti"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tow-thoughts-of-words-improve-reasoning-in","slug":"tow-thoughts-of-words-improve-reasoning-in","title":"ToW: Thoughts of Words Improve Reasoning in Large Language Models","date":"2024-10-21","arxiv_id":"2410.16235","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-graph-neural-networks-with-large","slug":"explaining-graph-neural-networks-with-large","title":"Explaining Graph Neural Networks with Large Language Models: A Counterfactual Perspective for Molecular Property Prediction","date":"2024-10-19","arxiv_id":"2410.15165","repositories_listed":1,"syntology":null},{"url":"/paper/paths-over-graph-knowledge-graph-enpowered","slug":"paths-over-graph-knowledge-graph-enpowered","title":"Paths-over-Graph: Knowledge Graph Empowered Large Language Model Reasoning","date":"2024-10-18","arxiv_id":"2410.14211","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/paths-over-graph-knowledge-graph-enpowered#ran","syntology_url":"https://syntology.ai/paper/2410.14211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14211"}},"official":null}},{"url":"/paper/rag-confusionqa-a-benchmark-for-evaluating","slug":"rag-confusionqa-a-benchmark-for-evaluating","title":"ELOQ: Resources for Enhancing LLM Detection of Out-of-Scope Questions","date":"2024-10-18","arxiv_id":"2410.14567","repositories_listed":1,"syntology":null},{"url":"/paper/from-single-to-multi-how-llms-hallucinate-in","slug":"from-single-to-multi-how-llms-hallucinate-in","title":"From Single to Multi: How LLMs Hallucinate in Multi-Document Summarization","date":"2024-10-17","arxiv_id":"2410.13961","repositories_listed":1,"syntology":null},{"url":"/paper/mcqg-srefine-multiple-choice-question","slug":"mcqg-srefine-multiple-choice-question","title":"MCQG-SRefine: Multiple Choice Question Generation and Evaluation with Iterative Self-Critique, Correction, and Comparison Feedback","date":"2024-10-17","arxiv_id":"2410.13191","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-hallucinations-in-large-vision-2","slug":"mitigating-hallucinations-in-large-vision-2","title":"Mitigating Hallucinations in Large Vision-Language Models via Summary-Guided Decoding","date":"2024-10-17","arxiv_id":"2410.13321","repositories_listed":1,"syntology":null},{"url":"/paper/a-claim-decomposition-benchmark-for-long-form","slug":"a-claim-decomposition-benchmark-for-long-form","title":"A Claim Decomposition Benchmark for Long-form Answer Verification","date":"2024-10-16","arxiv_id":"2410.12558","repositories_listed":1,"syntology":null},{"url":"/paper/graph-constrained-reasoning-faithful","slug":"graph-constrained-reasoning-faithful","title":"Graph-constrained Reasoning: Faithful Reasoning on Knowledge Graphs with Large Language Models","date":"2024-10-16","arxiv_id":"2410.13080","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graph-constrained-reasoning-faithful#ran","syntology_url":"https://syntology.ai/paper/2410.13080","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13080"}},"official":{"repos":["RManLuo/graph-constrained-reasoning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mmed-rag-versatile-multimodal-rag-system-for","slug":"mmed-rag-versatile-multimodal-rag-system-for","title":"MMed-RAG: Versatile Multimodal RAG System for Medical Vision Language Models","date":"2024-10-16","arxiv_id":"2410.13085","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mmed-rag-versatile-multimodal-rag-system-for#ran","syntology_url":"https://syntology.ai/paper/2410.13085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13085"}},"official":{"repos":["richard-peng-xia/mmed-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/the-curse-of-multi-modalities-evaluating","slug":"the-curse-of-multi-modalities-evaluating","title":"The Curse of Multi-Modalities: Evaluating Hallucinations of Large Multimodal Models across Language, Visual, and Audio","date":"2024-10-16","arxiv_id":"2410.12787","repositories_listed":1,"syntology":null},{"url":"/paper/automatically-generating-visual-hallucination","slug":"automatically-generating-visual-hallucination","title":"Automatically Generating Visual Hallucination Test Cases for Multimodal Large Language Models","date":"2024-10-15","arxiv_id":"2410.11242","repositories_listed":1,"syntology":null},{"url":"/paper/mllm-can-see-dynamic-correction-decoding-for","slug":"mllm-can-see-dynamic-correction-decoding-for","title":"MLLM can see? Dynamic Correction Decoding for Hallucination Mitigation","date":"2024-10-15","arxiv_id":"2410.11779","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/mllm-can-see-dynamic-correction-decoding-for#ran","syntology_url":"https://syntology.ai/paper/2410.11779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11779"}},"official":{"repos":["zjunlp/Deco"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/search-engines-in-an-ai-era-the-false-promise","slug":"search-engines-in-an-ai-era-the-false-promise","title":"Search Engines in an AI Era: The False Promise of Factual and Verifiable Source-Cited Responses","date":"2024-10-15","arxiv_id":"2410.22349","repositories_listed":1,"syntology":null},{"url":"/paper/videoagent-self-improving-video-generation","slug":"videoagent-self-improving-video-generation","title":"VideoAgent: Self-Improving Video Generation","date":"2024-10-14","arxiv_id":"2410.10076","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":4,"n_no_contract":6,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 4 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videoagent-self-improving-video-generation#ran","syntology_url":"https://syntology.ai/paper/2410.10076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10076"}},"official":{"repos":["video-as-agent/videoagent"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/longhalqa-long-context-hallucination","slug":"longhalqa-long-context-hallucination","title":"LongHalQA: Long-Context Hallucination Evaluation for MultiModal Large Language Models","date":"2024-10-13","arxiv_id":"2410.09962","repositories_listed":1,"syntology":null},{"url":"/paper/a-methodology-for-evaluating-rag-systems-a","slug":"a-methodology-for-evaluating-rag-systems-a","title":"A Methodology for Evaluating RAG Systems: A Case Study On Configuration Dependency Validation","date":"2024-10-11","arxiv_id":"2410.08801","repositories_listed":1,"syntology":null},{"url":"/paper/verified-a-video-corpus-moment-retrieval","slug":"verified-a-video-corpus-moment-retrieval","title":"VERIFIED: A Video Corpus Moment Retrieval Benchmark for Fine-Grained Video Understanding","date":"2024-10-11","arxiv_id":"2410.08593","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-curriculum-expert-iteration-for","slug":"automatic-curriculum-expert-iteration-for","title":"Automatic Curriculum Expert Iteration for Reliable LLM Reasoning","date":"2024-10-10","arxiv_id":"2410.07627","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automatic-curriculum-expert-iteration-for#ran","syntology_url":"https://syntology.ai/paper/2410.07627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07627"}},"official":{"repos":["salesforceairesearch/auto-cei"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/onenet-a-fine-tuning-free-framework-for-few","slug":"onenet-a-fine-tuning-free-framework-for-few","title":"OneNet: A Fine-Tuning Free Framework for Few-Shot Entity Linking via Large Language Model Prompting","date":"2024-10-10","arxiv_id":"2410.07549","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/onenet-a-fine-tuning-free-framework-for-few#ran","syntology_url":"https://syntology.ai/paper/2410.07549","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07549"}},"official":{"repos":["laquabe/OneNet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/itergen-iterative-structured-llm-generation","slug":"itergen-iterative-structured-llm-generation","title":"IterGen: Iterative Semantic-aware Structured LLM Generation with Backtracking","date":"2024-10-09","arxiv_id":"2410.07295","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/itergen-iterative-structured-llm-generation#ran","syntology_url":"https://syntology.ai/paper/2410.07295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07295"}},"official":{"repos":["uiuc-arc/itergen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/utilize-the-flow-before-stepping-into-the","slug":"utilize-the-flow-before-stepping-into-the","title":"Utilize the Flow before Stepping into the Same River Twice: Certainty Represented Knowledge Flow for Refusal-Aware Instruction Tuning","date":"2024-10-09","arxiv_id":"2410.06913","repositories_listed":1,"syntology":null},{"url":"/paper/refir-grounding-large-restoration-models-with","slug":"refir-grounding-large-restoration-models-with","title":"ReFIR: Grounding Large Restoration Models with Retrieval Augmentation","date":"2024-10-08","arxiv_id":"2410.05601","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/refir-grounding-large-restoration-models-with#ran","syntology_url":"https://syntology.ai/paper/2410.05601","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05601"}},"official":{"repos":["csguoh/refir"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-modality-prior-induced","slug":"mitigating-modality-prior-induced","title":"Mitigating Modality Prior-Induced Hallucinations in Multimodal Large Language Models via Deciphering Attention Causality","date":"2024-10-07","arxiv_id":"2410.04780","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mitigating-modality-prior-induced#ran","syntology_url":"https://syntology.ai/paper/2410.04780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04780"}},"official":{"repos":["the-martyr/causalmm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/tubench-benchmarking-large-vision-language","slug":"tubench-benchmarking-large-vision-language","title":"TUBench: Benchmarking Large Vision-Language Models on Trustworthiness with Unanswerable Questions","date":"2024-10-05","arxiv_id":"2410.04107","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-and-mitigating-object","slug":"investigating-and-mitigating-object","title":"Investigating and Mitigating Object Hallucinations in Pretrained Vision-Language (CLIP) Models","date":"2024-10-04","arxiv_id":"2410.03176","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/investigating-and-mitigating-object#ran","syntology_url":"https://syntology.ai/paper/2410.03176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03176"}},"official":{"repos":["yufang-liu/clip_hallucination"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/look-twice-before-you-answer-memory-space","slug":"look-twice-before-you-answer-memory-space","title":"Look Twice Before You Answer: Memory-Space Visual Retracing for Hallucination Mitigation in Multimodal Large Language Models","date":"2024-10-04","arxiv_id":"2410.03577","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/look-twice-before-you-answer-memory-space#ran","syntology_url":"https://syntology.ai/paper/2410.03577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03577"}},"official":{"repos":["1zhou-Wang/MemVR"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/characterizing-context-influence-and","slug":"characterizing-context-influence-and","title":"Characterizing Context Influence and Hallucination in Summarization","date":"2024-10-03","arxiv_id":"2410.03026","repositories_listed":1,"syntology":null}],"record_sha256":"5185d2285114fa32a74953b3a564d5993c031252eda96a9af8116516ead84488","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}