{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/hallucination/papers/ran/1","list_of":"/task/hallucination","task":"Hallucination","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":3,"rows_per_page":100,"rows":[1,100],"of":276,"counts":{"archive_papers_tagged":1816,"with_a_code_link":752,"where_syntology_ran_a_sample":276,"not_listed_spam_title":0,"listed":1816,"listed_where_code_ran":276,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":240,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":240,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/hallucination/papers/ran/1","prev":null,"next":"/task/hallucination/papers/ran/2","papers":[{"url":"/paper/mitigating-object-hallucinations-via-sentence","slug":"mitigating-object-hallucinations-via-sentence","title":"Mitigating Object Hallucinations via Sentence-Level Early Intervention","date":"2025-07-16","arxiv_id":"2507.12455","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mitigating-object-hallucinations-via-sentence#ran","syntology_url":"https://syntology.ai/paper/2507.12455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.12455"}},"official":{"repos":["pspdada/SENTINEL"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uqlm-a-python-package-for-uncertainty","slug":"uqlm-a-python-package-for-uncertainty","title":"UQLM: A Python Package for Uncertainty Quantification in Large Language Models","date":"2025-07-08","arxiv_id":"2507.06196","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/uqlm-a-python-package-for-uncertainty#ran","syntology_url":"https://syntology.ai/paper/2507.06196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.06196"}},"official":{"repos":["cvs-health/uqlm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/discosg-towards-discourse-level-text-scene","slug":"discosg-towards-discourse-level-text-scene","title":"DiscoSG: Towards Discourse-Level Text Scene Graph Parsing through Iterative Graph Refinement","date":"2025-06-18","arxiv_id":"2506.15583","repositories_listed":2,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":17,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":16,"n_pointer_only":19,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 1 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/discosg-towards-discourse-level-text-scene#ran","syntology_url":"https://syntology.ai/paper/2506.15583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.15583"}},"official":{"repos":["shaoqlin/discosg"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":17,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/second-mitigating-perceptual-hallucination-in","slug":"second-mitigating-perceptual-hallucination-in","title":"SECOND: Mitigating Perceptual Hallucination in Vision-Language Models via Selective and Contrastive Decoding","date":"2025-06-10","arxiv_id":"2506.08391","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/second-mitigating-perceptual-hallucination-in#ran","syntology_url":"https://syntology.ai/paper/2506.08391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08391"}},"official":{"repos":["aidaslab/second"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/memoir-lifelong-model-editing-with-minimal","slug":"memoir-lifelong-model-editing-with-minimal","title":"MEMOIR: Lifelong Model Editing with Minimal Overwrite and Informed Retention for LLMs","date":"2025-06-09","arxiv_id":"2506.07899","repositories_listed":0,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/memoir-lifelong-model-editing-with-minimal#ran","syntology_url":"https://syntology.ai/paper/2506.07899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07899"}},"official":null}},{"url":"/paper/the-hallucination-dilemma-factuality-aware","slug":"the-hallucination-dilemma-factuality-aware","title":"The Hallucination Dilemma: Factuality-Aware Reinforcement Learning for Large Reasoning Models","date":"2025-05-30","arxiv_id":"2505.24630","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-hallucination-dilemma-factuality-aware#ran","syntology_url":"https://syntology.ai/paper/2505.24630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24630"}},"official":{"repos":["nusnlp/fspo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/causal-llava-causal-disentanglement-for","slug":"causal-llava-causal-disentanglement-for","title":"Causal-LLaVA: Causal Disentanglement for Mitigating Hallucination in Multimodal Large Language Models","date":"2025-05-26","arxiv_id":"2505.19474","repositories_listed":1,"syntology":{"n":19,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":11,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 11 unverified","sample_list":"/paper/causal-llava-causal-disentanglement-for#ran","syntology_url":"https://syntology.ai/paper/2505.19474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19474"}},"official":{"repos":["ignisavium/causal-llava"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":11,"ran_from_kinds":["official"]}}},{"url":"/paper/attention-you-vision-language-model-could-be","slug":"attention-you-vision-language-model-could-be","title":"Attention! You Vision Language Model Could Be Maliciously Manipulated","date":"2025-05-26","arxiv_id":"2505.19911","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attention-you-vision-language-model-could-be#ran","syntology_url":"https://syntology.ai/paper/2505.19911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19911"}},"official":null}},{"url":"/paper/removal-of-hallucination-on-hallucination","slug":"removal-of-hallucination-on-hallucination","title":"Removal of Hallucination on Hallucination: Debate-Augmented RAG","date":"2025-05-24","arxiv_id":"2505.18581","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/removal-of-hallucination-on-hallucination#ran","syntology_url":"https://syntology.ai/paper/2505.18581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18581"}},"official":{"repos":["huenao/debate-augmented-rag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/audiotrust-benchmarking-the-multifaceted","slug":"audiotrust-benchmarking-the-multifaceted","title":"AudioTrust: Benchmarking the Multifaceted Trustworthiness of Audio Large Language Models","date":"2025-05-22","arxiv_id":"2505.16211","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audiotrust-benchmarking-the-multifaceted#ran","syntology_url":"https://syntology.ai/paper/2505.16211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16211"}},"official":{"repos":["jusperlee/audiotrust"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepeyes-incentivizing-thinking-with-images","slug":"deepeyes-incentivizing-thinking-with-images","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14362","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deepeyes-incentivizing-thinking-with-images#ran","syntology_url":"https://syntology.ai/paper/2505.14362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14362"}},"official":{"repos":["visual-agent/deepeyes"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/pierce-the-mists-greet-the-sky-decipher","slug":"pierce-the-mists-greet-the-sky-decipher","title":"Pierce the Mists, Greet the Sky: Decipher Knowledge Overshadowing via Knowledge Circuit Analysis","date":"2025-05-20","arxiv_id":"2505.14406","repositories_listed":0,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pierce-the-mists-greet-the-sky-decipher#ran","syntology_url":"https://syntology.ai/paper/2505.14406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14406"}},"official":null}},{"url":"/paper/a-head-to-predict-and-a-head-to-question-pre","slug":"a-head-to-predict-and-a-head-to-question-pre","title":"A Head to Predict and a Head to Question: Pre-trained Uncertainty Quantification Heads for Hallucination Detection in LLM Outputs","date":"2025-05-13","arxiv_id":"2505.08200","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/a-head-to-predict-and-a-head-to-question-pre#ran","syntology_url":"https://syntology.ai/paper/2505.08200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.08200"}},"official":null}},{"url":"/paper/evolutionary-thoughts-integration-of-large","slug":"evolutionary-thoughts-integration-of-large","title":"Evolutionary thoughts: integration of large language models and evolutionary algorithms","date":"2025-05-09","arxiv_id":"2505.05756","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evolutionary-thoughts-integration-of-large#ran","syntology_url":"https://syntology.ai/paper/2505.05756","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05756"}},"official":{"repos":["ajjimeno/fast-evolutionary-evaluation","ajjimeno/list-data","ajjimeno/llm-gp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-llm-faithfulness-in-rag-with","slug":"benchmarking-llm-faithfulness-in-rag-with","title":"Benchmarking LLM Faithfulness in RAG with Evolving Leaderboards","date":"2025-05-07","arxiv_id":"2505.04847","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/benchmarking-llm-faithfulness-in-rag-with#ran","syntology_url":"https://syntology.ai/paper/2505.04847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.04847"}},"official":{"repos":["vectara/FaithJudge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generate-but-verify-reducing-hallucination-in","slug":"generate-but-verify-reducing-hallucination-in","title":"Generate, but Verify: Reducing Hallucination in Vision-Language Models with Retrospective Resampling","date":"2025-04-17","arxiv_id":"2504.13169","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":1,"n_instrument":9,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 9 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generate-but-verify-reducing-hallucination-in#ran","syntology_url":"https://syntology.ai/paper/2504.13169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13169"}},"official":{"repos":["tsunghan-wu/reverse_vlm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/hallushift-measuring-distribution-shifts","slug":"hallushift-measuring-distribution-shifts","title":"HalluShift: Measuring Distribution Shifts towards Hallucination Detection in LLMs","date":"2025-04-13","arxiv_id":"2504.09482","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hallushift-measuring-distribution-shifts#ran","syntology_url":"https://syntology.ai/paper/2504.09482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.09482"}},"official":{"repos":["sharanya-dasgupta001/hallushift"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-hallucination-detection-in-llms-via","slug":"robust-hallucination-detection-in-llms-via","title":"Robust Hallucination Detection in LLMs via Adaptive Token Selection","date":"2025-04-10","arxiv_id":"2504.07863","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/robust-hallucination-detection-in-llms-via#ran","syntology_url":"https://syntology.ai/paper/2504.07863","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07863"}},"official":null}},{"url":"/paper/hoigen-1m-a-large-scale-dataset-for-human","slug":"hoigen-1m-a-large-scale-dataset-for-human","title":"HOIGen-1M: A Large-scale Dataset for Human-Object Interaction Video Generation","date":"2025-03-31","arxiv_id":"2503.23715","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hoigen-1m-a-large-scale-dataset-for-human#ran","syntology_url":"https://syntology.ai/paper/2503.23715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23715"}},"official":null}},{"url":"/paper/better-wit-than-wealth-dynamic-parametric","slug":"better-wit-than-wealth-dynamic-parametric","title":"Dynamic Parametric Retrieval Augmented Generation for Test-time Knowledge Enhancement","date":"2025-03-31","arxiv_id":"2503.23895","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/better-wit-than-wealth-dynamic-parametric#ran","syntology_url":"https://syntology.ai/paper/2503.23895","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23895"}},"official":{"repos":["trae1oung/dyprag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rare-retrieval-augmented-reasoning-modeling","slug":"rare-retrieval-augmented-reasoning-modeling","title":"RARE: Retrieval-Augmented Reasoning Modeling","date":"2025-03-30","arxiv_id":"2503.23513","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rare-retrieval-augmented-reasoning-modeling#ran","syntology_url":"https://syntology.ai/paper/2503.23513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23513"}},"official":{"repos":["open-dataflow/rare"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tn-eval-rubric-and-evaluation-protocols-for","slug":"tn-eval-rubric-and-evaluation-protocols-for","title":"TN-Eval: Rubric and Evaluation Protocols for Measuring the Quality of Behavioral Therapy Notes","date":"2025-03-26","arxiv_id":"2503.20648","repositories_listed":0,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tn-eval-rubric-and-evaluation-protocols-for#ran","syntology_url":"https://syntology.ai/paper/2503.20648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20648"}},"official":null}},{"url":"/paper/cafe-unifying-representation-and-generation","slug":"cafe-unifying-representation-and-generation","title":"CAFe: Unifying Representation and Generation with Contrastive-Autoregressive Finetuning","date":"2025-03-25","arxiv_id":"2503.19900","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cafe-unifying-representation-and-generation#ran","syntology_url":"https://syntology.ai/paper/2503.19900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19900"}},"official":{"repos":["haoyu-bu/CAFe"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/learning-on-llm-output-signatures-for-gray","slug":"learning-on-llm-output-signatures-for-gray","title":"Learning on LLM Output Signatures for gray-box LLM Behavior Analysis","date":"2025-03-18","arxiv_id":"2503.14043","repositories_listed":1,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/learning-on-llm-output-signatures-for-gray#ran","syntology_url":"https://syntology.ai/paper/2503.14043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.14043"}},"official":{"repos":["barsguy/llm-output-signatures-network"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/vlrmbench-a-comprehensive-and-challenging","slug":"vlrmbench-a-comprehensive-and-challenging","title":"VLRMBench: A Comprehensive and Challenging Benchmark for Vision-Language Reward Models","date":"2025-03-10","arxiv_id":"2503.07478","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlrmbench-a-comprehensive-and-challenging#ran","syntology_url":"https://syntology.ai/paper/2503.07478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07478"}},"official":{"repos":["jcruan519/vlrmbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-hallucinations-in-large-vision-4","slug":"mitigating-hallucinations-in-large-vision-4","title":"Mitigating Hallucinations in Large Vision-Language Models by Adaptively Constraining Information Flow","date":"2025-02-28","arxiv_id":"2502.20750","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mitigating-hallucinations-in-large-vision-4#ran","syntology_url":"https://syntology.ai/paper/2502.20750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20750"}},"official":{"repos":["jiaqi5598/adavib"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/treecut-a-synthetic-unanswerable-math-word","slug":"treecut-a-synthetic-unanswerable-math-word","title":"TreeCut: A Synthetic Unanswerable Math Word Problem Dataset for LLM Hallucination Evaluation","date":"2025-02-19","arxiv_id":"2502.13442","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/treecut-a-synthetic-unanswerable-math-word#ran","syntology_url":"https://syntology.ai/paper/2502.13442","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13442"}},"official":{"repos":["j-bagel/treecut-math"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-conformal-abstention-policies-for","slug":"learning-conformal-abstention-policies-for","title":"Learning Conformal Abstention Policies for Adaptive Risk Management in Large Language and Vision-Language Models","date":"2025-02-08","arxiv_id":"2502.06884","repositories_listed":1,"syntology":{"n":16,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/learning-conformal-abstention-policies-for#ran","syntology_url":"https://syntology.ai/paper/2502.06884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06884"}},"official":{"repos":["sinatayebati/vlm-uncertainty"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/videorope-what-makes-for-good-video-rotary","slug":"videorope-what-makes-for-good-video-rotary","title":"VideoRoPE: What Makes for Good Video Rotary Position Embedding?","date":"2025-02-07","arxiv_id":"2502.05173","repositories_listed":1,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":13,"n_instrument":3,"n_unverified":1,"n_honours":3,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/videorope-what-makes-for-good-video-rotary#ran","syntology_url":"https://syntology.ai/paper/2502.05173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05173"}},"official":{"repos":["wiselnn570/videorope"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-hidden-life-of-tokens-reducing","slug":"the-hidden-life-of-tokens-reducing","title":"The Hidden Life of Tokens: Reducing Hallucination of Large Vision-Language Models via Visual Information Steering","date":"2025-02-05","arxiv_id":"2502.03628","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-hidden-life-of-tokens-reducing#ran","syntology_url":"https://syntology.ai/paper/2502.03628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.03628"}},"official":{"repos":["LzVv123456/VISTA"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/differentially-private-steering-for-large","slug":"differentially-private-steering-for-large","title":"Differentially Private Steering for Large Language Model Alignment","date":"2025-01-30","arxiv_id":"2501.18532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/differentially-private-steering-for-large#ran","syntology_url":"https://syntology.ai/paper/2501.18532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.18532"}},"official":{"repos":["ukplab/iclr2025-psa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chip-cross-modal-hierarchical-direct","slug":"chip-cross-modal-hierarchical-direct","title":"CHiP: Cross-modal Hierarchical Direct Preference Optimization for Multimodal LLMs","date":"2025-01-28","arxiv_id":"2501.16629","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/chip-cross-modal-hierarchical-direct#ran","syntology_url":"https://syntology.ai/paper/2501.16629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.16629"}},"official":{"repos":["lvugai/chip"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-hallucinations-in-large-vision-3","slug":"mitigating-hallucinations-in-large-vision-3","title":"Mitigating Hallucinations in Large Vision-Language Models via DPO: On-Policy Data Hold the Key","date":"2025-01-16","arxiv_id":"2501.09695","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":16,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mitigating-hallucinations-in-large-vision-3#ran","syntology_url":"https://syntology.ai/paper/2501.09695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.09695"}},"official":{"repos":["zhyang2226/opa-dpo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/tarsier2-advancing-large-vision-language","slug":"tarsier2-advancing-large-vision-language","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","date":"2025-01-14","arxiv_id":"2501.07888","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier2-advancing-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2501.07888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.07888"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/ecbench-can-multi-modal-foundation-models","slug":"ecbench-can-multi-modal-foundation-models","title":"ECBench: Can Multi-modal Foundation Models Understand the Egocentric World? A Holistic Embodied Cognition Benchmark","date":"2025-01-09","arxiv_id":"2501.05031","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ecbench-can-multi-modal-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2501.05031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05031"}},"official":{"repos":["rh-dang/ecbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-large-language-models-for-1","slug":"harnessing-large-language-models-for-1","title":"Harnessing Large Language Models for Knowledge Graph Question Answering via Adaptive Multi-Aspect Retrieval-Augmentation","date":"2024-12-24","arxiv_id":"2412.18537","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/harnessing-large-language-models-for-1#ran","syntology_url":"https://syntology.ai/paper/2412.18537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18537"}},"official":{"repos":["Applied-Machine-Learning-Lab/AMAR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/citebart-learning-to-generate-citations-for","slug":"citebart-learning-to-generate-citations-for","title":"CiteBART: Learning to Generate Citations for Local Citation Recommendation","date":"2024-12-23","arxiv_id":"2412.17534","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/citebart-learning-to-generate-citations-for#ran","syntology_url":"https://syntology.ai/paper/2412.17534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.17534"}},"official":{"repos":["eyclk/citationrecommendation"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/granite-guardian","slug":"granite-guardian","title":"Granite Guardian","date":"2024-12-10","arxiv_id":"2412.07724","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/granite-guardian#ran","syntology_url":"https://syntology.ai/paper/2412.07724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07724"}},"official":{"repos":["ibm-granite/granite-guardian"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/expanding-performance-boundaries-of-open","slug":"expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","arxiv_id":"2412.05271","repositories_listed":1,"syntology":{"n":9,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/expanding-performance-boundaries-of-open#ran","syntology_url":"https://syntology.ai/paper/2412.05271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05271"}},"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/can-llms-be-good-graph-judger-for-knowledge","slug":"can-llms-be-good-graph-judger-for-knowledge","title":"Can LLMs be Good Graph Judge for Knowledge Graph Construction?","date":"2024-11-26","arxiv_id":"2411.17388","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-llms-be-good-graph-judger-for-knowledge#ran","syntology_url":"https://syntology.ai/paper/2411.17388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17388"}},"official":{"repos":["hhy-huang/graphjudge"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/atomr-atomic-operator-empowered-large","slug":"atomr-atomic-operator-empowered-large","title":"AtomR: Atomic Operator-Empowered Large Language Models for Heterogeneous Knowledge Reasoning","date":"2024-11-25","arxiv_id":"2411.16495","repositories_listed":1,"syntology":{"n":13,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":13,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/atomr-atomic-operator-empowered-large#ran","syntology_url":"https://syntology.ai/paper/2411.16495","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16495"}},"official":{"repos":["THU-KEG/AtomR"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/devils-in-middle-layers-of-large-vision","slug":"devils-in-middle-layers-of-large-vision","title":"Devils in Middle Layers of Large Vision-Language Models: Interpreting, Detecting and Mitigating Object Hallucinations via Attention Lens","date":"2024-11-23","arxiv_id":"2411.16724","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/devils-in-middle-layers-of-large-vision#ran","syntology_url":"https://syntology.ai/paper/2411.16724","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16724"}},"official":{"repos":["zhangqijiang07/middle_layers_indicating_hallucinations"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-uncertainty-detecting-hallucination-in","slug":"vl-uncertainty-detecting-hallucination-in","title":"VL-Uncertainty: Detecting Hallucination in Large Vision-Language Model via Uncertainty Estimation","date":"2024-11-18","arxiv_id":"2411.11919","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":3,"n_honours":3,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 3 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vl-uncertainty-detecting-hallucination-in#ran","syntology_url":"https://syntology.ai/paper/2411.11919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.11919"}},"official":null}},{"url":"/paper/assistrag-boosting-the-potential-of-large","slug":"assistrag-boosting-the-potential-of-large","title":"AssistRAG: Boosting the Potential of Large Language Models with an Intelligent Information Assistant","date":"2024-11-11","arxiv_id":"2411.06805","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":1,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"10 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/assistrag-boosting-the-potential-of-large#ran","syntology_url":"https://syntology.ai/paper/2411.06805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.06805"}},"official":{"repos":["smallporridge/assistrag"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":1,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/v-dpo-mitigating-hallucination-in-large","slug":"v-dpo-mitigating-hallucination-in-large","title":"V-DPO: Mitigating Hallucination in Large Vision Language Models via Vision-Guided Direct Preference Optimization","date":"2024-11-05","arxiv_id":"2411.02712","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/v-dpo-mitigating-hallucination-in-large#ran","syntology_url":"https://syntology.ai/paper/2411.02712","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02712"}},"official":{"repos":["yuxixie/v-dpo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-multimodal-retrieval-augmented","slug":"benchmarking-multimodal-retrieval-augmented","title":"Benchmarking Multimodal Retrieval Augmented Generation with Dynamic VQA Dataset and Self-adaptive Planning Agent","date":"2024-11-05","arxiv_id":"2411.02937","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/benchmarking-multimodal-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2411.02937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02937"}},"official":{"repos":["alibaba-nlp/omnisearch"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/timesuite-improving-mllms-for-long-video","slug":"timesuite-improving-mllms-for-long-video","title":"TimeSuite: Improving MLLMs for Long Video Understanding via Grounded Tuning","date":"2024-10-25","arxiv_id":"2410.19702","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/timesuite-improving-mllms-for-long-video#ran","syntology_url":"https://syntology.ai/paper/2410.19702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.19702"}},"official":null}},{"url":"/paper/avhbench-a-cross-modal-hallucination","slug":"avhbench-a-cross-modal-hallucination","title":"AVHBench: A Cross-Modal Hallucination Benchmark for Audio-Visual Large Language Models","date":"2024-10-23","arxiv_id":"2410.18325","repositories_listed":0,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/avhbench-a-cross-modal-hallucination#ran","syntology_url":"https://syntology.ai/paper/2410.18325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18325"}},"official":null}},{"url":"/paper/navigating-noisy-feedback-enhancing","slug":"navigating-noisy-feedback-enhancing","title":"Navigating Noisy Feedback: Enhancing Reinforcement Learning with Error-Prone Language Models","date":"2024-10-22","arxiv_id":"2410.17389","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/navigating-noisy-feedback-enhancing#ran","syntology_url":"https://syntology.ai/paper/2410.17389","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17389"}},"official":{"repos":["sy-shi/RLAIF_ScoreDiff"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/reducing-hallucinations-in-vision-language","slug":"reducing-hallucinations-in-vision-language","title":"Reducing Hallucinations in Vision-Language Models via Latent Space Steering","date":"2024-10-21","arxiv_id":"2410.15778","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reducing-hallucinations-in-vision-language#ran","syntology_url":"https://syntology.ai/paper/2410.15778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15778"}},"official":{"repos":["shengliu66/vti"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/can-knowledge-editing-really-correct","slug":"can-knowledge-editing-really-correct","title":"Can Knowledge Editing Really Correct Hallucinations?","date":"2024-10-21","arxiv_id":"2410.16251","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":14,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/can-knowledge-editing-really-correct#ran","syntology_url":"https://syntology.ai/paper/2410.16251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16251"}},"official":{"repos":["llm-editing/HalluEditBench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/paths-over-graph-knowledge-graph-enpowered","slug":"paths-over-graph-knowledge-graph-enpowered","title":"Paths-over-Graph: Knowledge Graph Empowered Large Language Model Reasoning","date":"2024-10-18","arxiv_id":"2410.14211","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/paths-over-graph-knowledge-graph-enpowered#ran","syntology_url":"https://syntology.ai/paper/2410.14211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14211"}},"official":null}},{"url":"/paper/graph-constrained-reasoning-faithful","slug":"graph-constrained-reasoning-faithful","title":"Graph-constrained Reasoning: Faithful Reasoning on Knowledge Graphs with Large Language Models","date":"2024-10-16","arxiv_id":"2410.13080","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graph-constrained-reasoning-faithful#ran","syntology_url":"https://syntology.ai/paper/2410.13080","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13080"}},"official":{"repos":["RManLuo/graph-constrained-reasoning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mmed-rag-versatile-multimodal-rag-system-for","slug":"mmed-rag-versatile-multimodal-rag-system-for","title":"MMed-RAG: Versatile Multimodal RAG System for Medical Vision Language Models","date":"2024-10-16","arxiv_id":"2410.13085","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mmed-rag-versatile-multimodal-rag-system-for#ran","syntology_url":"https://syntology.ai/paper/2410.13085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13085"}},"official":{"repos":["richard-peng-xia/mmed-rag"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/videoagent-self-improving-video-generation","slug":"videoagent-self-improving-video-generation","title":"VideoAgent: Self-Improving Video Generation","date":"2024-10-14","arxiv_id":"2410.10076","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":4,"n_no_contract":6,"n_pointer_only":2,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 4 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/videoagent-self-improving-video-generation#ran","syntology_url":"https://syntology.ai/paper/2410.10076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10076"}},"official":{"repos":["video-as-agent/videoagent"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vlfeedback-a-large-scale-ai-feedback-dataset","slug":"vlfeedback-a-large-scale-ai-feedback-dataset","title":"VLFeedback: A Large-Scale AI Feedback Dataset for Large Vision-Language Models Alignment","date":"2024-10-12","arxiv_id":"2410.09421","repositories_listed":0,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vlfeedback-a-large-scale-ai-feedback-dataset#ran","syntology_url":"https://syntology.ai/paper/2410.09421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09421"}},"official":null}},{"url":"/paper/onenet-a-fine-tuning-free-framework-for-few","slug":"onenet-a-fine-tuning-free-framework-for-few","title":"OneNet: A Fine-Tuning Free Framework for Few-Shot Entity Linking via Large Language Model Prompting","date":"2024-10-10","arxiv_id":"2410.07549","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/onenet-a-fine-tuning-free-framework-for-few#ran","syntology_url":"https://syntology.ai/paper/2410.07549","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07549"}},"official":{"repos":["laquabe/OneNet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-curriculum-expert-iteration-for","slug":"automatic-curriculum-expert-iteration-for","title":"Automatic Curriculum Expert Iteration for Reliable LLM Reasoning","date":"2024-10-10","arxiv_id":"2410.07627","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automatic-curriculum-expert-iteration-for#ran","syntology_url":"https://syntology.ai/paper/2410.07627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07627"}},"official":{"repos":["salesforceairesearch/auto-cei"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/embodied-agent-interface-benchmarking-llms","slug":"embodied-agent-interface-benchmarking-llms","title":"Embodied Agent Interface: Benchmarking LLMs for Embodied Decision Making","date":"2024-10-09","arxiv_id":"2410.07166","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/embodied-agent-interface-benchmarking-llms#ran","syntology_url":"https://syntology.ai/paper/2410.07166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07166"}},"official":{"repos":["embodied-agent-eval/embodied-agent-eval","embodied-agent-interface/embodied-agent-interface","embodied-agent-eval/embodied-agent-eval.github.io"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/itergen-iterative-structured-llm-generation","slug":"itergen-iterative-structured-llm-generation","title":"IterGen: Iterative Semantic-aware Structured LLM Generation with Backtracking","date":"2024-10-09","arxiv_id":"2410.07295","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/itergen-iterative-structured-llm-generation#ran","syntology_url":"https://syntology.ai/paper/2410.07295","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07295"}},"official":{"repos":["uiuc-arc/itergen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/refir-grounding-large-restoration-models-with","slug":"refir-grounding-large-restoration-models-with","title":"ReFIR: Grounding Large Restoration Models with Retrieval Augmentation","date":"2024-10-08","arxiv_id":"2410.05601","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/refir-grounding-large-restoration-models-with#ran","syntology_url":"https://syntology.ai/paper/2410.05601","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05601"}},"official":{"repos":["csguoh/refir"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-modality-prior-induced","slug":"mitigating-modality-prior-induced","title":"Mitigating Modality Prior-Induced Hallucinations in Multimodal Large Language Models via Deciphering Attention Causality","date":"2024-10-07","arxiv_id":"2410.04780","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mitigating-modality-prior-induced#ran","syntology_url":"https://syntology.ai/paper/2410.04780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04780"}},"official":{"repos":["the-martyr/causalmm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/differential-transformer","slug":"differential-transformer","title":"Differential Transformer","date":"2024-10-07","arxiv_id":"2410.05258","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/differential-transformer#ran","syntology_url":"https://syntology.ai/paper/2410.05258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.05258"}},"official":{"repos":["microsoft/unilm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/investigating-and-mitigating-object","slug":"investigating-and-mitigating-object","title":"Investigating and Mitigating Object Hallucinations in Pretrained Vision-Language (CLIP) Models","date":"2024-10-04","arxiv_id":"2410.03176","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/investigating-and-mitigating-object#ran","syntology_url":"https://syntology.ai/paper/2410.03176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03176"}},"official":{"repos":["yufang-liu/clip_hallucination"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/look-twice-before-you-answer-memory-space","slug":"look-twice-before-you-answer-memory-space","title":"Look Twice Before You Answer: Memory-Space Visual Retracing for Hallucination Mitigation in Multimodal Large Language Models","date":"2024-10-04","arxiv_id":"2410.03577","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/look-twice-before-you-answer-memory-space#ran","syntology_url":"https://syntology.ai/paper/2410.03577","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03577"}},"official":{"repos":["1zhou-Wang/MemVR"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bordirlines-a-dataset-for-evaluating-cross","slug":"bordirlines-a-dataset-for-evaluating-cross","title":"BordIRlines: A Dataset for Evaluating Cross-lingual Retrieval-Augmented Generation","date":"2024-10-02","arxiv_id":"2410.01171","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/bordirlines-a-dataset-for-evaluating-cross#ran","syntology_url":"https://syntology.ai/paper/2410.01171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01171"}},"official":{"repos":["manestay/bordirlines"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-hallucinations-in-practical-code","slug":"llm-hallucinations-in-practical-code","title":"LLM Hallucinations in Practical Code Generation: Phenomena, Mechanism, and Mitigation","date":"2024-09-30","arxiv_id":"2409.20550","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llm-hallucinations-in-practical-code#ran","syntology_url":"https://syntology.ai/paper/2409.20550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20550"}},"official":{"repos":["deepsoftwareanalytics/llmcodinghallucination"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/haloscope-harnessing-unlabeled-llm","slug":"haloscope-harnessing-unlabeled-llm","title":"HaloScope: Harnessing Unlabeled LLM Generations for Hallucination Detection","date":"2024-09-26","arxiv_id":"2409.17504","repositories_listed":0,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":10,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":16,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/haloscope-harnessing-unlabeled-llm#ran","syntology_url":"https://syntology.ai/paper/2409.17504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17504"}},"official":null}},{"url":"/paper/eventhallusion-diagnosing-event","slug":"eventhallusion-diagnosing-event","title":"EventHallusion: Diagnosing Event Hallucinations in Video LLMs","date":"2024-09-25","arxiv_id":"2409.16597","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eventhallusion-diagnosing-event#ran","syntology_url":"https://syntology.ai/paper/2409.16597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16597"}},"official":{"repos":["stevetich/eventhallusion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/controlling-risk-of-retrieval-augmented","slug":"controlling-risk-of-retrieval-augmented","title":"Controlling Risk of Retrieval-augmented Generation: A Counterfactual Prompting Framework","date":"2024-09-24","arxiv_id":"2409.16146","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/controlling-risk-of-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2409.16146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16146"}},"official":{"repos":["ict-bigdatalab/rc-rag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/journeybench-a-challenging-one-stop-vision","slug":"journeybench-a-challenging-one-stop-vision","title":"JourneyBench: A Challenging One-Stop Vision-Language Understanding Benchmark of Generated Images","date":"2024-09-19","arxiv_id":"2409.12953","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/journeybench-a-challenging-one-stop-vision#ran","syntology_url":"https://syntology.ai/paper/2409.12953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12953"}},"official":{"repos":["journeybench/journeybench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/thames-an-end-to-end-tool-for-hallucination","slug":"thames-an-end-to-end-tool-for-hallucination","title":"THaMES: An End-to-End Tool for Hallucination Mitigation and Evaluation in Large Language Models","date":"2024-09-17","arxiv_id":"2409.11353","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/thames-an-end-to-end-tool-for-hallucination#ran","syntology_url":"https://syntology.ai/paper/2409.11353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.11353"}},"official":{"repos":["holistic-ai/THaMES"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/trustworthiness-in-retrieval-augmented","slug":"trustworthiness-in-retrieval-augmented","title":"Trustworthiness in Retrieval-Augmented Generation Systems: A Survey","date":"2024-09-16","arxiv_id":"2409.10102","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trustworthiness-in-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2409.10102","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.10102"}},"official":{"repos":["smallporridge/trustworthyrag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/confidence-estimation-for-llm-based-dialogue","slug":"confidence-estimation-for-llm-based-dialogue","title":"Confidence Estimation for LLM-Based Dialogue State Tracking","date":"2024-09-15","arxiv_id":"2409.09629","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/confidence-estimation-for-llm-based-dialogue#ran","syntology_url":"https://syntology.ai/paper/2409.09629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.09629"}},"official":{"repos":["jennycs0830/confidence_score_dst"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-mod-making-llava-tiny-via-moe-knowledge","slug":"llava-mod-making-llava-tiny-via-moe-knowledge","title":"LLaVA-MoD: Making LLaVA Tiny via MoE Knowledge Distillation","date":"2024-08-28","arxiv_id":"2408.15881","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/llava-mod-making-llava-tiny-via-moe-knowledge#ran","syntology_url":"https://syntology.ai/paper/2408.15881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15881"}},"official":{"repos":["shufangxun/llava-mod"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/convis-contrastive-decoding-with","slug":"convis-contrastive-decoding-with","title":"ConVis: Contrastive Decoding with Hallucination Visualization for Mitigating Hallucinations in Multimodal Large Language Models","date":"2024-08-25","arxiv_id":"2408.13906","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/convis-contrastive-decoding-with#ran","syntology_url":"https://syntology.ai/paper/2408.13906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.13906"}},"official":{"repos":["yejipark-m/convis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rovrm-a-robust-visual-reward-model-optimized","slug":"rovrm-a-robust-visual-reward-model-optimized","title":"RoVRM: A Robust Visual Reward Model Optimized via Auxiliary Textual Preference Data","date":"2024-08-22","arxiv_id":"2408.12109","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rovrm-a-robust-visual-reward-model-optimized#ran","syntology_url":"https://syntology.ai/paper/2408.12109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12109"}},"official":{"repos":["wangclnlp/vision-llm-alignment"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-factuality-in-large-language-models","slug":"improving-factuality-in-large-language-models","title":"Improving Factuality in Large Language Models via Decoding-Time Hallucinatory and Truthful Comparators","date":"2024-08-22","arxiv_id":"2408.12325","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-factuality-in-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2408.12325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12325"}},"official":{"repos":["ydk122024/cdt"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reefknot-a-comprehensive-benchmark-for","slug":"reefknot-a-comprehensive-benchmark-for","title":"Reefknot: A Comprehensive Benchmark for Relation Hallucination Evaluation, Analysis and Mitigation in Multimodal Large Language Models","date":"2024-08-18","arxiv_id":"2408.09429","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reefknot-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2408.09429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09429"}},"official":{"repos":["JackChen-seu/Reefknot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/order-matters-in-hallucination-reasoning","slug":"order-matters-in-hallucination-reasoning","title":"Order Matters in Hallucination: Reasoning Order as Benchmark and Reflexive Prompting for Large-Language-Models","date":"2024-08-09","arxiv_id":"2408.05093","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/order-matters-in-hallucination-reasoning#ran","syntology_url":"https://syntology.ai/paper/2408.05093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.05093"}},"official":{"repos":["xiezikai/reflexiveprompting"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-02032","slug":"2408-02032","title":"Self-Introspective Decoding: Alleviating Hallucinations for Large Vision-Language Models","date":"2024-08-04","arxiv_id":"2408.02032","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-02032#ran","syntology_url":"https://syntology.ai/paper/2408.02032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.02032"}},"official":{"repos":["huofushuo/SID"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/2408-01800","slug":"2408-01800","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","date":"2024-08-03","arxiv_id":"2408.01800","repositories_listed":2,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/2408-01800#ran","syntology_url":"https://syntology.ai/paper/2408.01800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01800"}},"official":{"repos":["OpenBMB/MiniCPM-o"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/paying-more-attention-to-image-a-training","slug":"paying-more-attention-to-image-a-training","title":"Paying More Attention to Image: A Training-Free Method for Alleviating Hallucination in LVLMs","date":"2024-07-31","arxiv_id":"2407.21771","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/paying-more-attention-to-image-a-training#ran","syntology_url":"https://syntology.ai/paper/2407.21771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21771"}},"official":null}},{"url":"/paper/automated-review-generation-method-based-on","slug":"automated-review-generation-method-based-on","title":"Automated Review Generation Method Based on Large Language Models","date":"2024-07-30","arxiv_id":"2407.20906","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/automated-review-generation-method-based-on#ran","syntology_url":"https://syntology.ai/paper/2407.20906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.20906"}},"official":{"repos":["tju-ecat-ai/automaticreviewgeneration"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-llm-s-cognition-via-structurization","slug":"enhancing-llm-s-cognition-via-structurization","title":"Enhancing LLM's Cognition via Structurization","date":"2024-07-23","arxiv_id":"2407.16434","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-llm-s-cognition-via-structurization#ran","syntology_url":"https://syntology.ai/paper/2407.16434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16434"}},"official":{"repos":["alibaba/struxgpt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/maven-fact-a-large-scale-event-factuality","slug":"maven-fact-a-large-scale-event-factuality","title":"MAVEN-Fact: A Large-scale Event Factuality Detection Dataset","date":"2024-07-22","arxiv_id":"2407.15352","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/maven-fact-a-large-scale-event-factuality#ran","syntology_url":"https://syntology.ai/paper/2407.15352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15352"}},"official":{"repos":["lcy2723/maven-fact","THU-KEG/MAVEN-FACT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-dynamics-of-llm-finetuning","slug":"learning-dynamics-of-llm-finetuning","title":"Learning Dynamics of LLM Finetuning","date":"2024-07-15","arxiv_id":"2407.10490","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-dynamics-of-llm-finetuning#ran","syntology_url":"https://syntology.ai/paper/2407.10490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10490"}},"official":{"repos":["joshua-ren/learning_dynamics_llm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/lookback-lens-detecting-and-mitigating","slug":"lookback-lens-detecting-and-mitigating","title":"Lookback Lens: Detecting and Mitigating Contextual Hallucinations in Large Language Models Using Only Attention Maps","date":"2024-07-09","arxiv_id":"2407.07071","repositories_listed":1,"syntology":{"n":20,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":20,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/lookback-lens-detecting-and-mitigating#ran","syntology_url":"https://syntology.ai/paper/2407.07071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07071"}},"official":{"repos":["voidism/lookback-lens"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-object-hallucination-in-vision-language","slug":"multi-object-hallucination-in-vision-language","title":"Multi-Object Hallucination in Vision-Language Models","date":"2024-07-08","arxiv_id":"2407.06192","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-object-hallucination-in-vision-language#ran","syntology_url":"https://syntology.ai/paper/2407.06192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06192"}},"official":{"repos":["sled-group/moh"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-hallucination-detection-through","slug":"enhancing-hallucination-detection-through","title":"Enhancing Hallucination Detection through Perturbation-Based Synthetic Data Generation in System Responses","date":"2024-07-07","arxiv_id":"2407.05474","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-hallucination-detection-through#ran","syntology_url":"https://syntology.ai/paper/2407.05474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05474"}},"official":{"repos":["asappresearch/halugen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/anah-v2-scaling-analytical-hallucination","slug":"anah-v2-scaling-analytical-hallucination","title":"ANAH-v2: Scaling Analytical Hallucination Annotation of Large Language Models","date":"2024-07-05","arxiv_id":"2407.04693","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/anah-v2-scaling-analytical-hallucination#ran","syntology_url":"https://syntology.ai/paper/2407.04693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04693"}},"official":{"repos":["open-compass/anah"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mj-bench-is-your-multimodal-reward-model","slug":"mj-bench-is-your-multimodal-reward-model","title":"MJ-Bench: Is Your Multimodal Reward Model Really a Good Judge for Text-to-Image Generation?","date":"2024-07-05","arxiv_id":"2407.04842","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mj-bench-is-your-multimodal-reward-model#ran","syntology_url":"https://syntology.ai/paper/2407.04842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04842"}},"official":{"repos":["MJ-Bench/MJ-Bench"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-internal-states-reveal-hallucination-risk","slug":"llm-internal-states-reveal-hallucination-risk","title":"LLM Internal States Reveal Hallucination Risk Faced With a Query","date":"2024-07-03","arxiv_id":"2407.03282","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm-internal-states-reveal-hallucination-risk#ran","syntology_url":"https://syntology.ai/paper/2407.03282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03282"}},"official":{"repos":["ziweiji/Internal_States_Reveal_Hallucination"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-involuntary-truth","slug":"large-language-models-are-involuntary-truth","title":"Large Language Models Are Involuntary Truth-Tellers: Exploiting Fallacy Failure for Jailbreak Attacks","date":"2024-07-01","arxiv_id":"2407.00869","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/large-language-models-are-involuntary-truth#ran","syntology_url":"https://syntology.ai/paper/2407.00869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00869"}},"official":{"repos":["Yue-LLM-Pit/FFA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/grapharena-benchmarking-large-language-models","slug":"grapharena-benchmarking-large-language-models","title":"GraphArena: Benchmarking Large Language Models on Graph Computational Problems","date":"2024-06-29","arxiv_id":"2407.00379","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/grapharena-benchmarking-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2407.00379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00379"}},"official":{"repos":["squareroot3/grapharena"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/toolbehonest-a-multi-level-hallucination","slug":"toolbehonest-a-multi-level-hallucination","title":"ToolBeHonest: A Multi-level Hallucination Diagnostic Benchmark for Tool-Augmented Large Language Models","date":"2024-06-28","arxiv_id":"2406.20015","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/toolbehonest-a-multi-level-hallucination#ran","syntology_url":"https://syntology.ai/paper/2406.20015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20015"}},"official":{"repos":["toolbehonest/toolbehonest"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/understand-what-llm-needs-dual-preference","slug":"understand-what-llm-needs-dual-preference","title":"Understand What LLM Needs: Dual Preference Alignment for Retrieval-Augmented Generation","date":"2024-06-26","arxiv_id":"2406.18676","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/understand-what-llm-needs-dual-preference#ran","syntology_url":"https://syntology.ai/paper/2406.18676","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18676"}},"official":{"repos":["dongguanting/dpa-rag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-and-analyzing-relationship","slug":"evaluating-and-analyzing-relationship","title":"Evaluating and Analyzing Relationship Hallucinations in Large Vision-Language Models","date":"2024-06-24","arxiv_id":"2406.16449","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-and-analyzing-relationship#ran","syntology_url":"https://syntology.ai/paper/2406.16449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16449"}},"official":{"repos":["mrwu-mac/R-Bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/semantic-entropy-probes-robust-and-cheap","slug":"semantic-entropy-probes-robust-and-cheap","title":"Semantic Entropy Probes: Robust and Cheap Hallucination Detection in LLMs","date":"2024-06-22","arxiv_id":"2406.15927","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/semantic-entropy-probes-robust-and-cheap#ran","syntology_url":"https://syntology.ai/paper/2406.15927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15927"}},"official":null}},{"url":"/paper/evaluating-rag-fusion-with-ragelo-an","slug":"evaluating-rag-fusion-with-ragelo-an","title":"Evaluating RAG-Fusion with RAGElo: an Automated Elo-based Framework","date":"2024-06-20","arxiv_id":"2406.14783","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-rag-fusion-with-ragelo-an#ran","syntology_url":"https://syntology.ai/paper/2406.14783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14783"}},"official":{"repos":["zetaalphavector/ragelo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"45b69bf6a79e265bcb083817287c36a127bd9e353ce482983d45b334030df539","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}