{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/hallucination/papers/2","list_of":"/task/hallucination","task":"Hallucination","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":19,"rows_per_page":100,"rows":[101,200],"of":1816,"counts":{"archive_papers_tagged":1816,"with_a_code_link":752,"where_syntology_ran_a_sample":276,"not_listed_spam_title":0,"listed":1816,"listed_where_code_ran":276,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":240,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":240,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/hallucination","prev":"/task/hallucination","next":"/task/hallucination/papers/3","papers":[{"url":"/paper/finmme-benchmark-dataset-for-financial-multi","slug":"finmme-benchmark-dataset-for-financial-multi","title":"FinMME: Benchmark Dataset for Financial Multi-Modal Reasoning Evaluation","date":"2025-05-30","arxiv_id":"2505.24714","repositories_listed":1,"syntology":null},{"url":"/paper/llm-inference-enhanced-by-external-knowledge","slug":"llm-inference-enhanced-by-external-knowledge","title":"LLM Inference Enhanced by External Knowledge: A Survey","date":"2025-05-30","arxiv_id":"2505.24377","repositories_listed":1,"syntology":null},{"url":"/paper/the-hallucination-dilemma-factuality-aware","slug":"the-hallucination-dilemma-factuality-aware","title":"The Hallucination Dilemma: Factuality-Aware Reinforcement Learning for Large Reasoning Models","date":"2025-05-30","arxiv_id":"2505.24630","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-hallucination-dilemma-factuality-aware#ran","syntology_url":"https://syntology.ai/paper/2505.24630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24630"}},"official":{"repos":["nusnlp/fspo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mmboundary-advancing-mllm-knowledge-boundary","slug":"mmboundary-advancing-mllm-knowledge-boundary","title":"MMBoundary: Advancing MLLM Knowledge Boundary Awareness through Reasoning Step Confidence Calibration","date":"2025-05-29","arxiv_id":"2505.23224","repositories_listed":1,"syntology":null},{"url":"/paper/qwen-look-again-guiding-vision-language","slug":"qwen-look-again-guiding-vision-language","title":"Qwen Look Again: Guiding Vision-Language Reasoning Models to Re-attention Visual Information","date":"2025-05-29","arxiv_id":"2505.23558","repositories_listed":1,"syntology":null},{"url":"/paper/cognibench-a-legal-inspired-framework-and","slug":"cognibench-a-legal-inspired-framework-and","title":"CogniBench: A Legal-inspired Framework and Dataset for Assessing Cognitive Faithfulness of Large Language Models","date":"2025-05-27","arxiv_id":"2505.20767","repositories_listed":1,"syntology":null},{"url":"/paper/causal-llava-causal-disentanglement-for","slug":"causal-llava-causal-disentanglement-for","title":"Causal-LLaVA: Causal Disentanglement for Mitigating Hallucination in Multimodal Large Language Models","date":"2025-05-26","arxiv_id":"2505.19474","repositories_listed":1,"syntology":{"n":19,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":11,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/causal-llava-causal-disentanglement-for#ran","syntology_url":"https://syntology.ai/paper/2505.19474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19474"}},"official":{"repos":["ignisavium/causal-llava"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":11,"ran_from_kinds":["official"]}}},{"url":"/paper/error-typing-for-smarter-rewards-improving","slug":"error-typing-for-smarter-rewards-improving","title":"Error Typing for Smarter Rewards: Improving Process Reward Models with Error-Aware Hierarchical Supervision","date":"2025-05-26","arxiv_id":"2505.19706","repositories_listed":1,"syntology":null},{"url":"/paper/omni-r1-reinforcement-learning-for-omnimodal","slug":"omni-r1-reinforcement-learning-for-omnimodal","title":"Omni-R1: Reinforcement Learning for Omnimodal Reasoning via Two-System Collaboration","date":"2025-05-26","arxiv_id":"2505.20256","repositories_listed":1,"syntology":null},{"url":"/paper/r3-rag-learning-step-by-step-reasoning-and","slug":"r3-rag-learning-step-by-step-reasoning-and","title":"R3-RAG: Learning Step-by-Step Reasoning and Retrieval for LLMs via Reinforcement Learning","date":"2025-05-26","arxiv_id":"2505.23794","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-visual-contrastive-decoding-to","slug":"retrieval-visual-contrastive-decoding-to","title":"Retrieval Visual Contrastive Decoding to Mitigate Object Hallucinations in Large Vision-Language Models","date":"2025-05-26","arxiv_id":"2505.20569","repositories_listed":1,"syntology":null},{"url":"/paper/cchall-a-novel-benchmark-for-joint-cross","slug":"cchall-a-novel-benchmark-for-joint-cross","title":"CCHall: A Novel Benchmark for Joint Cross-Lingual and Cross-Modal Hallucinations Detection in Large Language Models","date":"2025-05-25","arxiv_id":"2505.19108","repositories_listed":1,"syntology":null},{"url":"/paper/medscore-factuality-evaluation-of-free-form","slug":"medscore-factuality-evaluation-of-free-form","title":"MedScore: Factuality Evaluation of Free-Form Medical Answers","date":"2025-05-24","arxiv_id":"2505.18452","repositories_listed":1,"syntology":null},{"url":"/paper/removal-of-hallucination-on-hallucination","slug":"removal-of-hallucination-on-hallucination","title":"Removal of Hallucination on Hallucination: Debate-Augmented RAG","date":"2025-05-24","arxiv_id":"2505.18581","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/removal-of-hallucination-on-hallucination#ran","syntology_url":"https://syntology.ai/paper/2505.18581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18581"}},"official":{"repos":["huenao/debate-augmented-rag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/keepitsimple-at-semeval-2025-task-3-llm","slug":"keepitsimple-at-semeval-2025-task-3-llm","title":"keepitsimple at SemEval-2025 Task 3: LLM-Uncertainty based Approach for Multilingual Hallucination Span Detection","date":"2025-05-23","arxiv_id":"2505.17485","repositories_listed":1,"syntology":null},{"url":"/paper/audiotrust-benchmarking-the-multifaceted","slug":"audiotrust-benchmarking-the-multifaceted","title":"AudioTrust: Benchmarking the Multifaceted Trustworthiness of Audio Large Language Models","date":"2025-05-22","arxiv_id":"2505.16211","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiotrust-benchmarking-the-multifaceted#ran","syntology_url":"https://syntology.ai/paper/2505.16211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16211"}},"official":{"repos":["jusperlee/audiotrust"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-hallucinations-in-vision-language","slug":"mitigating-hallucinations-in-vision-language","title":"Mitigating Hallucinations in Vision-Language Models through Image-Guided Head Suppression","date":"2025-05-22","arxiv_id":"2505.16411","repositories_listed":1,"syntology":null},{"url":"/paper/walk-retrieve-simple-yet-effective-zero-shot","slug":"walk-retrieve-simple-yet-effective-zero-shot","title":"Walk&Retrieve: Simple Yet Effective Zero-shot Retrieval-Augmented Generation via Knowledge Graph Walks","date":"2025-05-22","arxiv_id":"2505.16849","repositories_listed":1,"syntology":null},{"url":"/paper/deepeyes-incentivizing-thinking-with-images","slug":"deepeyes-incentivizing-thinking-with-images","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14362","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deepeyes-incentivizing-thinking-with-images#ran","syntology_url":"https://syntology.ai/paper/2505.14362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14362"}},"official":{"repos":["visual-agent/deepeyes"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/multihal-multilingual-dataset-for-knowledge","slug":"multihal-multilingual-dataset-for-knowledge","title":"MultiHal: Multilingual Dataset for Knowledge-Graph Grounded Evaluation of LLM Hallucinations","date":"2025-05-20","arxiv_id":"2505.14101","repositories_listed":1,"syntology":null},{"url":"/paper/toward-reliable-biomedical-hypothesis","slug":"toward-reliable-biomedical-hypothesis","title":"Toward Reliable Biomedical Hypothesis Generation: Evaluating Truthfulness and Hallucination in Large Language Models","date":"2025-05-20","arxiv_id":"2505.14599","repositories_listed":1,"syntology":null},{"url":"/paper/know-or-not-a-library-for-evaluating-out-of","slug":"know-or-not-a-library-for-evaluating-out-of","title":"Know Or Not: a library for evaluating out-of-knowledge base robustness","date":"2025-05-19","arxiv_id":"2505.13545","repositories_listed":1,"syntology":null},{"url":"/paper/llm-based-query-expansion-fails-for","slug":"llm-based-query-expansion-fails-for","title":"LLM-based Query Expansion Fails for Unfamiliar and Ambiguous Queries","date":"2025-05-19","arxiv_id":"2505.12694","repositories_listed":1,"syntology":null},{"url":"/paper/mixture-of-decoding-an-attention-inspired","slug":"mixture-of-decoding-an-attention-inspired","title":"Mixture of Decoding: An Attention-Inspired Adaptive Decoding Strategy to Mitigate Hallucinations in Large Vision-Language Models","date":"2025-05-17","arxiv_id":"2505.17061","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11365","slug":"2505-11365","title":"Phare: A Safety Probe for Large Language Models","date":"2025-05-16","arxiv_id":"2505.11365","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11405","slug":"2505-11405","title":"EmotionHallucer: Evaluating Emotion Hallucinations in Multimodal Large Language Models","date":"2025-05-16","arxiv_id":"2505.11405","repositories_listed":1,"syntology":null},{"url":"/paper/finetune-rag-fine-tuning-language-models-to","slug":"finetune-rag-fine-tuning-language-models-to","title":"Finetune-RAG: Fine-Tuning Language Models to Resist Hallucination in Retrieval-Augmented Generation","date":"2025-05-16","arxiv_id":"2505.10792","repositories_listed":1,"syntology":null},{"url":"/paper/do-rag-a-domain-specific-qa-framework-using","slug":"do-rag-a-domain-specific-qa-framework-using","title":"DO-RAG: A Domain-Specific QA Framework Using Knowledge Graph-Enhanced Retrieval-Augmented Generation","date":"2025-05-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-head-to-predict-and-a-head-to-question-pre","slug":"a-head-to-predict-and-a-head-to-question-pre","title":"A Head to Predict and a Head to Question: Pre-trained Uncertainty Quantification Heads for Hallucination Detection in LLM Outputs","date":"2025-05-13","arxiv_id":"2505.08200","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-head-to-predict-and-a-head-to-question-pre#ran","syntology_url":"https://syntology.ai/paper/2505.08200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.08200"}},"official":null}},{"url":"/paper/prioritizing-image-related-tokens-enhances","slug":"prioritizing-image-related-tokens-enhances","title":"Prioritizing Image-Related Tokens Enhances Vision-Language Pre-Training","date":"2025-05-13","arxiv_id":"2505.08971","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-survival-modeling-in-the-age-of","slug":"multimodal-survival-modeling-in-the-age-of","title":"Multimodal Survival Modeling in the Age of Foundation Models","date":"2025-05-12","arxiv_id":"2505.07683","repositories_listed":1,"syntology":null},{"url":"/paper/hallucination-aware-multimodal-benchmark-for","slug":"hallucination-aware-multimodal-benchmark-for","title":"Hallucination-Aware Multimodal Benchmark for Gastrointestinal Image Analysis with Large Vision-Language Models","date":"2025-05-11","arxiv_id":"2505.07001","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-llm-faithfulness-in-rag-with","slug":"benchmarking-llm-faithfulness-in-rag-with","title":"Benchmarking LLM Faithfulness in RAG with Evolving Leaderboards","date":"2025-05-07","arxiv_id":"2505.04847","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/benchmarking-llm-faithfulness-in-rag-with#ran","syntology_url":"https://syntology.ai/paper/2505.04847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.04847"}},"official":{"repos":["vectara/FaithJudge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/invoke-interfaces-only-when-needed-adaptive","slug":"invoke-interfaces-only-when-needed-adaptive","title":"Invoke Interfaces Only When Needed: Adaptive Invocation for Large Language Models in Question Answering","date":"2025-05-05","arxiv_id":"2505.02311","repositories_listed":1,"syntology":null},{"url":"/paper/ucsc-at-semeval-2025-task-3-context-models","slug":"ucsc-at-semeval-2025-task-3-context-models","title":"UCSC at SemEval-2025 Task 3: Context, Models and Prompt Optimization for Automated Hallucination Detection in LLM Output","date":"2025-05-05","arxiv_id":"2505.03030","repositories_listed":1,"syntology":null},{"url":"/paper/regression-is-all-you-need-for-medical-image","slug":"regression-is-all-you-need-for-medical-image","title":"Regression is all you need for medical image translation","date":"2025-05-04","arxiv_id":"2505.02048","repositories_listed":1,"syntology":null},{"url":"/paper/videohallu-evaluating-and-mitigating-multi","slug":"videohallu-evaluating-and-mitigating-multi","title":"VideoHallu: Evaluating and Mitigating Multi-modal Hallucinations on Synthetic Video Understanding","date":"2025-05-02","arxiv_id":"2505.01481","repositories_listed":1,"syntology":null},{"url":"/paper/smallplan-leverage-small-language-models-for","slug":"smallplan-leverage-small-language-models-for","title":"SmallPlan: Leverage Small Language Models for Sequential Path Planning with Simulation-Powered, LLM-Guided Distillation","date":"2025-05-01","arxiv_id":"2505.00831","repositories_listed":1,"syntology":null},{"url":"/paper/antidote-a-unified-framework-for-mitigating","slug":"antidote-a-unified-framework-for-mitigating","title":"Antidote: A Unified Framework for Mitigating LVLM Hallucinations in Counterfactual Presupposition and Object Perception","date":"2025-04-29","arxiv_id":"2504.20468","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-quantification-for-language","slug":"uncertainty-quantification-for-language","title":"Uncertainty Quantification for Language Models: A Suite of Black-Box, White-Box, LLM Judge, and Ensemble Scorers","date":"2025-04-27","arxiv_id":"2504.19254","repositories_listed":1,"syntology":null},{"url":"/paper/dyfo-a-training-free-dynamic-focus-visual","slug":"dyfo-a-training-free-dynamic-focus-visual","title":"DyFo: A Training-Free Dynamic Focus Visual Search for Enhancing LMMs in Fine-Grained Visual Understanding","date":"2025-04-21","arxiv_id":"2504.14920","repositories_listed":1,"syntology":null},{"url":"/paper/analyzing-llms-knowledge-boundary-cognition","slug":"analyzing-llms-knowledge-boundary-cognition","title":"Analyzing LLMs' Knowledge Boundary Cognition Across Languages Through the Lens of Internal Representations","date":"2025-04-18","arxiv_id":"2504.13816","repositories_listed":1,"syntology":null},{"url":"/paper/generate-but-verify-reducing-hallucination-in","slug":"generate-but-verify-reducing-hallucination-in","title":"Generate, but Verify: Reducing Hallucination in Vision-Language Models with Retrospective Resampling","date":"2025-04-17","arxiv_id":"2504.13169","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":1,"n_instrument":9,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 9 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generate-but-verify-reducing-hallucination-in#ran","syntology_url":"https://syntology.ai/paper/2504.13169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13169"}},"official":{"repos":["tsunghan-wu/reverse_vlm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/vistadpo-video-hierarchical-spatial-temporal","slug":"vistadpo-video-hierarchical-spatial-temporal","title":"VistaDPO: Video Hierarchical Spatial-Temporal Direct Preference Optimization for Large Video Models","date":"2025-04-17","arxiv_id":"2504.13122","repositories_listed":1,"syntology":null},{"url":"/paper/why-and-how-llms-hallucinate-connecting-the","slug":"why-and-how-llms-hallucinate-connecting-the","title":"Why and How LLMs Hallucinate: Connecting the Dots with Subsequence Associations","date":"2025-04-17","arxiv_id":"2504.12691","repositories_listed":1,"syntology":null},{"url":"/paper/semeval-2025-task-3-mu-shroom-the","slug":"semeval-2025-task-3-mu-shroom-the","title":"SemEval-2025 Task 3: Mu-SHROOM, the Multilingual Shared Task on Hallucinations and Related Observable Overgeneration Mistakes","date":"2025-04-16","arxiv_id":"2504.11975","repositories_listed":1,"syntology":null},{"url":"/paper/embodiedagent-a-scalable-hierarchical","slug":"embodiedagent-a-scalable-hierarchical","title":"EmbodiedAgent: A Scalable Hierarchical Approach to Overcome Practical Challenge in Multi-Robot Control","date":"2025-04-14","arxiv_id":"2504.10030","repositories_listed":1,"syntology":null},{"url":"/paper/the-mirage-of-performance-gains-why","slug":"the-mirage-of-performance-gains-why","title":"The Mirage of Performance Gains: Why Contrastive Decoding Fails to Address Multimodal Hallucination","date":"2025-04-14","arxiv_id":"2504.10020","repositories_listed":1,"syntology":null},{"url":"/paper/hallushift-measuring-distribution-shifts","slug":"hallushift-measuring-distribution-shifts","title":"HalluShift: Measuring Distribution Shifts towards Hallucination Detection in LLMs","date":"2025-04-13","arxiv_id":"2504.09482","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hallushift-measuring-distribution-shifts#ran","syntology_url":"https://syntology.ai/paper/2504.09482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.09482"}},"official":{"repos":["sharanya-dasgupta001/hallushift"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-other-side-of-the-coin-exploring-fairness","slug":"the-other-side-of-the-coin-exploring-fairness","title":"The Other Side of the Coin: Exploring Fairness in Retrieval-Augmented Generation","date":"2025-04-11","arxiv_id":"2504.12323","repositories_listed":1,"syntology":null},{"url":"/paper/learning-fine-grained-domain-generalization","slug":"learning-fine-grained-domain-generalization","title":"Learning Fine-grained Domain Generalization via Hyperbolic State Space Hallucination","date":"2025-04-10","arxiv_id":"2504.08020","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-dynamic-clustering-based-document","slug":"efficient-dynamic-clustering-based-document","title":"Efficient Dynamic Clustering-Based Document Compression for Retrieval-Augmented-Generation","date":"2025-04-04","arxiv_id":"2504.03165","repositories_listed":1,"syntology":null},{"url":"/paper/noise-augmented-fine-tuning-for-mitigating","slug":"noise-augmented-fine-tuning-for-mitigating","title":"Noise Augmented Fine Tuning for Mitigating Hallucinations in Large Language Models","date":"2025-04-04","arxiv_id":"2504.03302","repositories_listed":1,"syntology":null},{"url":"/paper/catch-me-if-you-search-when-contextual-web","slug":"catch-me-if-you-search-when-contextual-web","title":"Catch Me if You Search: When Contextual Web Search Results Affect the Detection of Hallucinations","date":"2025-04-01","arxiv_id":"2504.01153","repositories_listed":1,"syntology":null},{"url":"/paper/better-wit-than-wealth-dynamic-parametric","slug":"better-wit-than-wealth-dynamic-parametric","title":"Dynamic Parametric Retrieval Augmented Generation for Test-time Knowledge Enhancement","date":"2025-03-31","arxiv_id":"2503.23895","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/better-wit-than-wealth-dynamic-parametric#ran","syntology_url":"https://syntology.ai/paper/2503.23895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23895"}},"official":{"repos":["trae1oung/dyprag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rare-retrieval-augmented-reasoning-modeling","slug":"rare-retrieval-augmented-reasoning-modeling","title":"RARE: Retrieval-Augmented Reasoning Modeling","date":"2025-03-30","arxiv_id":"2503.23513","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rare-retrieval-augmented-reasoning-modeling#ran","syntology_url":"https://syntology.ai/paper/2503.23513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23513"}},"official":{"repos":["open-dataflow/rare"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gapo-learning-preferential-prompt-through","slug":"gapo-learning-preferential-prompt-through","title":"GAPO: Learning Preferential Prompt through Generative Adversarial Policy Optimization","date":"2025-03-26","arxiv_id":"2503.20194","repositories_listed":1,"syntology":null},{"url":"/paper/cafe-unifying-representation-and-generation","slug":"cafe-unifying-representation-and-generation","title":"CAFe: Unifying Representation and Generation with Contrastive-Autoregressive Finetuning","date":"2025-03-25","arxiv_id":"2503.19900","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cafe-unifying-representation-and-generation#ran","syntology_url":"https://syntology.ai/paper/2503.19900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19900"}},"official":{"repos":["haoyu-bu/CAFe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/exploring-hallucination-of-large-multimodal","slug":"exploring-hallucination-of-large-multimodal","title":"Exploring Hallucination of Large Multimodal Models in Video Understanding: Benchmark, Analysis and Mitigation","date":"2025-03-25","arxiv_id":"2503.19622","repositories_listed":1,"syntology":null},{"url":"/paper/lrsclip-a-vision-language-foundation-model","slug":"lrsclip-a-vision-language-foundation-model","title":"LRSCLIP: A Vision-Language Foundation Model for Aligning Remote Sensing Image with Longer Text","date":"2025-03-25","arxiv_id":"2503.19311","repositories_listed":1,"syntology":null},{"url":"/paper/geobenchx-benchmarking-llms-for-multistep","slug":"geobenchx-benchmarking-llms-for-multistep","title":"GeoBenchX: Benchmarking LLMs for Multistep Geospatial Tasks","date":"2025-03-23","arxiv_id":"2503.18129","repositories_listed":1,"syntology":null},{"url":"/paper/prodehaze-prompting-diffusion-models-toward","slug":"prodehaze-prompting-diffusion-models-toward","title":"ProDehaze: Prompting Diffusion Models Toward Faithful Image Dehazing","date":"2025-03-21","arxiv_id":"2503.17488","repositories_listed":1,"syntology":null},{"url":"/paper/towards-lighter-and-robust-evaluation-for","slug":"towards-lighter-and-robust-evaluation-for","title":"Towards Lighter and Robust Evaluation for Retrieval Augmented Generation","date":"2025-03-20","arxiv_id":"2503.16161","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/towards-lighter-and-robust-evaluation-for#ran","syntology_url":"https://syntology.ai/paper/2503.16161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16161"}},"official":{"repos":["razvanip13/towards_lighter_and_robust_evaluation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/learning-on-llm-output-signatures-for-gray","slug":"learning-on-llm-output-signatures-for-gray","title":"Learning on LLM Output Signatures for gray-box LLM Behavior Analysis","date":"2025-03-18","arxiv_id":"2503.14043","repositories_listed":1,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/learning-on-llm-output-signatures-for-gray#ran","syntology_url":"https://syntology.ai/paper/2503.14043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.14043"}},"official":{"repos":["barsguy/llm-output-signatures-network"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/clearsight-visual-signal-enhancement-for","slug":"clearsight-visual-signal-enhancement-for","title":"ClearSight: Visual Signal Enhancement for Object Hallucination Mitigation in Multimodal Large language Models","date":"2025-03-17","arxiv_id":"2503.13107","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/clearsight-visual-signal-enhancement-for#ran","syntology_url":"https://syntology.ai/paper/2503.13107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13107"}},"official":{"repos":["ustc-hyin/ClearSight"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/grounded-chain-of-thought-for-multimodal","slug":"grounded-chain-of-thought-for-multimodal","title":"Grounded Chain-of-Thought for Multimodal Large Language Models","date":"2025-03-17","arxiv_id":"2503.12799","repositories_listed":1,"syntology":null},{"url":"/paper/hicd-hallucination-inducing-via-attention","slug":"hicd-hallucination-inducing-via-attention","title":"HICD: Hallucination-Inducing via Attention Dispersion for Contrastive Decoding to Mitigate Hallucinations in Large Language Models","date":"2025-03-17","arxiv_id":"2503.12908","repositories_listed":1,"syntology":null},{"url":"/paper/aistorian-lets-ai-be-a-historian-a-kg-powered","slug":"aistorian-lets-ai-be-a-historian-a-kg-powered","title":"AIstorian lets AI be a historian: A KG-powered multi-agent system for accurate biography generation","date":"2025-03-14","arxiv_id":"2503.11346","repositories_listed":1,"syntology":null},{"url":"/paper/prompt-injection-detection-and-mitigation-via","slug":"prompt-injection-detection-and-mitigation-via","title":"Prompt Injection Detection and Mitigation via AI Multi-Agent NLP Frameworks","date":"2025-03-14","arxiv_id":"2503.11517","repositories_listed":1,"syntology":null},{"url":"/paper/truthprint-mitigating-lvlm-object","slug":"truthprint-mitigating-lvlm-object","title":"TruthPrInt: Mitigating LVLM Object Hallucination Via Latent Truthful-Guided Pre-Intervention","date":"2025-03-13","arxiv_id":"2503.10602","repositories_listed":1,"syntology":null},{"url":"/paper/conversational-gold-evaluating-personalized","slug":"conversational-gold-evaluating-personalized","title":"Conversational Gold: Evaluating Personalized Conversational Search System using Gold Nuggets","date":"2025-03-12","arxiv_id":"2503.09902","repositories_listed":1,"syntology":null},{"url":"/paper/nvp-hri-zero-shot-natural-voice-and-posture","slug":"nvp-hri-zero-shot-natural-voice-and-posture","title":"NVP-HRI: Zero Shot Natural Voice and Posture-based Human-Robot Interaction via Large Language Model","date":"2025-03-12","arxiv_id":"2503.09335","repositories_listed":1,"syntology":null},{"url":"/paper/vlrmbench-a-comprehensive-and-challenging","slug":"vlrmbench-a-comprehensive-and-challenging","title":"VLRMBench: A Comprehensive and Challenging Benchmark for Vision-Language Reward Models","date":"2025-03-10","arxiv_id":"2503.07478","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlrmbench-a-comprehensive-and-challenging#ran","syntology_url":"https://syntology.ai/paper/2503.07478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07478"}},"official":{"repos":["jcruan519/vlrmbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/treble-counterfactual-vlms-a-causal-approach","slug":"treble-counterfactual-vlms-a-causal-approach","title":"Treble Counterfactual VLMs: A Causal Approach to Hallucination","date":"2025-03-08","arxiv_id":"2503.06169","repositories_listed":1,"syntology":null},{"url":"/paper/attentive-reasoning-queries-a-systematic","slug":"attentive-reasoning-queries-a-systematic","title":"Attentive Reasoning Queries: A Systematic Method for Optimizing Instruction-Following in Large Language Models","date":"2025-03-05","arxiv_id":"2503.03669","repositories_listed":1,"syntology":null},{"url":"/paper/mcitebench-a-benchmark-for-multimodal","slug":"mcitebench-a-benchmark-for-multimodal","title":"MCiteBench: A Multimodal Benchmark for Generating Text with Citations","date":"2025-03-04","arxiv_id":"2503.02589","repositories_listed":1,"syntology":null},{"url":"/paper/shakespearean-sparks-the-dance-of","slug":"shakespearean-sparks-the-dance-of","title":"Shakespearean Sparks: The Dance of Hallucination and Creativity in LLMs' Decoding Layers","date":"2025-03-04","arxiv_id":"2503.02851","repositories_listed":1,"syntology":null},{"url":"/paper/wmnav-integrating-vision-language-models-into","slug":"wmnav-integrating-vision-language-models-into","title":"WMNav: Integrating Vision-Language Models into World Models for Object Goal Navigation","date":"2025-03-04","arxiv_id":"2503.02247","repositories_listed":1,"syntology":null},{"url":"/paper/2503-01670","slug":"2503-01670","title":"Evaluating LLMs' Assessment of Mixed-Context Hallucination Through the Lens of Summarization","date":"2025-03-03","arxiv_id":"2503.01670","repositories_listed":1,"syntology":null},{"url":"/paper/ncl-uor-at-semeval-2025-task-3-detecting","slug":"ncl-uor-at-semeval-2025-task-3-detecting","title":"NCL-UoR at SemEval-2025 Task 3: Detecting Multilingual Hallucination and Related Observable Overgeneration Text Spans with Modified RefChecker and Modified SeflCheckGPT","date":"2025-03-02","arxiv_id":"2503.01921","repositories_listed":1,"syntology":null},{"url":"/paper/u-niah-unified-rag-and-llm-evaluation-for","slug":"u-niah-unified-rag-and-llm-evaluation-for","title":"U-NIAH: Unified RAG and LLM Evaluation for Long Context Needle-In-A-Haystack","date":"2025-03-01","arxiv_id":"2503.00353","repositories_listed":1,"syntology":null},{"url":"/paper/medhalltune-an-instruction-tuning-benchmark","slug":"medhalltune-an-instruction-tuning-benchmark","title":"MedHallTune: An Instruction-Tuning Benchmark for Mitigating Medical Hallucination in Vision-Language Models","date":"2025-02-28","arxiv_id":"2502.20780","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-hallucinations-in-large-vision-4","slug":"mitigating-hallucinations-in-large-vision-4","title":"Mitigating Hallucinations in Large Vision-Language Models by Adaptively Constraining Information Flow","date":"2025-02-28","arxiv_id":"2502.20750","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mitigating-hallucinations-in-large-vision-4#ran","syntology_url":"https://syntology.ai/paper/2502.20750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20750"}},"official":{"repos":["jiaqi5598/adavib"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-general-visual-linguistic-face-1","slug":"towards-general-visual-linguistic-face-1","title":"Towards General Visual-Linguistic Face Forgery Detection(V2)","date":"2025-02-28","arxiv_id":"2502.20698","repositories_listed":1,"syntology":null},{"url":"/paper/one-for-more-continual-diffusion-model-for","slug":"one-for-more-continual-diffusion-model-for","title":"One-for-More: Continual Diffusion Model for Anomaly Detection","date":"2025-02-27","arxiv_id":"2502.19848","repositories_listed":1,"syntology":null},{"url":"/paper/proapo-progressively-automatic-prompt","slug":"proapo-progressively-automatic-prompt","title":"ProAPO: Progressively Automatic Prompt Optimization for Visual Classification","date":"2025-02-27","arxiv_id":"2502.19844","repositories_listed":1,"syntology":null},{"url":"/paper/vision-encoders-already-know-what-they-see","slug":"vision-encoders-already-know-what-they-see","title":"Vision-Encoders (Already) Know What They See: Mitigating Object Hallucination via Simple Fine-Grained CLIPScore","date":"2025-02-27","arxiv_id":"2502.20034","repositories_listed":1,"syntology":null},{"url":"/paper/medical-hallucinations-in-foundation-models","slug":"medical-hallucinations-in-foundation-models","title":"Medical Hallucinations in Foundation Models and Their Impact on Healthcare","date":"2025-02-26","arxiv_id":"2503.05777","repositories_listed":1,"syntology":null},{"url":"/paper/verdict-a-library-for-scaling-judge-time","slug":"verdict-a-library-for-scaling-judge-time","title":"Verdict: A Library for Scaling Judge-Time Compute","date":"2025-02-25","arxiv_id":"2502.18018","repositories_listed":1,"syntology":null},{"url":"/paper/hallucination-detection-in-llms-using","slug":"hallucination-detection-in-llms-using","title":"Hallucination Detection in LLMs Using Spectral Features of Attention Maps","date":"2025-02-24","arxiv_id":"2502.17598","repositories_listed":1,"syntology":null},{"url":"/paper/llm-qe-improving-query-expansion-by-aligning","slug":"llm-qe-improving-query-expansion-by-aligning","title":"LLM-QE: Improving Query Expansion by Aligning Large Language Models with Ranking Preferences","date":"2025-02-24","arxiv_id":"2502.17057","repositories_listed":1,"syntology":null},{"url":"/paper/pip-kag-mitigating-knowledge-conflicts-in","slug":"pip-kag-mitigating-knowledge-conflicts-in","title":"PIP-KAG: Mitigating Knowledge Conflicts in Knowledge-Augmented Generation via Parametric Pruning","date":"2025-02-21","arxiv_id":"2502.15543","repositories_listed":1,"syntology":null},{"url":"/paper/segsub-evaluating-robustness-to-knowledge","slug":"segsub-evaluating-robustness-to-knowledge","title":"SegSub: Evaluating Robustness to Knowledge Conflicts and Hallucinations in Vision-Language Models","date":"2025-02-19","arxiv_id":"2502.14908","repositories_listed":1,"syntology":null},{"url":"/paper/treecut-a-synthetic-unanswerable-math-word","slug":"treecut-a-synthetic-unanswerable-math-word","title":"TreeCut: A Synthetic Unanswerable Math Word Problem Dataset for LLM Hallucination Evaluation","date":"2025-02-19","arxiv_id":"2502.13442","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treecut-a-synthetic-unanswerable-math-word#ran","syntology_url":"https://syntology.ai/paper/2502.13442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13442"}},"official":{"repos":["j-bagel/treecut-math"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-much-do-llms-hallucinate-across-languages","slug":"how-much-do-llms-hallucinate-across-languages","title":"How Much Do LLMs Hallucinate across Languages? On Multilingual Estimation of LLM Hallucination in the Wild","date":"2025-02-18","arxiv_id":"2502.12769","repositories_listed":1,"syntology":null},{"url":"/paper/r2-kg-general-purpose-dual-agent-framework","slug":"r2-kg-general-purpose-dual-agent-framework","title":"R2-KG: General-Purpose Dual-Agent Framework for Reliable Reasoning on Knowledge Graphs","date":"2025-02-18","arxiv_id":"2502.12767","repositories_listed":1,"syntology":null},{"url":"/paper/unveiling-the-magic-of-code-reasoning-through","slug":"unveiling-the-magic-of-code-reasoning-through","title":"Unveiling the Magic of Code Reasoning through Hypothesis Decomposition and Amendment","date":"2025-02-17","arxiv_id":"2502.13170","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/unveiling-the-magic-of-code-reasoning-through#ran","syntology_url":"https://syntology.ai/paper/2502.13170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13170"}},"official":{"repos":["tntwow/code_reasoning"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/automated-hypothesis-validation-with-agentic","slug":"automated-hypothesis-validation-with-agentic","title":"Automated Hypothesis Validation with Agentic Sequential Falsifications","date":"2025-02-14","arxiv_id":"2502.09858","repositories_listed":1,"syntology":null},{"url":"/paper/elevating-legal-llm-responses-harnessing","slug":"elevating-legal-llm-responses-harnessing","title":"Elevating Legal LLM Responses: Harnessing Trainable Logical Structures and Semantic Knowledge with Legal Reasoning","date":"2025-02-11","arxiv_id":"2502.07912","repositories_listed":1,"syntology":null},{"url":"/paper/hallucination-monofacts-and-miscalibration-an","slug":"hallucination-monofacts-and-miscalibration-an","title":"Hallucination, Monofacts, and Miscalibration: An Empirical Investigation","date":"2025-02-11","arxiv_id":"2502.08666","repositories_listed":1,"syntology":null}],"record_sha256":"144c925d0ab0bd85a2ee03626c5da5c083c47bbc893097a7a43b34cdd845879a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}