{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/hallucination/papers/4","list_of":"/task/hallucination","task":"Hallucination","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":19,"rows_per_page":100,"rows":[301,400],"of":1816,"counts":{"archive_papers_tagged":1816,"with_a_code_link":752,"where_syntology_ran_a_sample":276,"not_listed_spam_title":0,"listed":1816,"listed_where_code_ran":276,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":240,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":240,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/hallucination","prev":"/task/hallucination/papers/3","next":"/task/hallucination/papers/5","papers":[{"url":"/paper/crispo-multi-aspect-critique-suggestion","slug":"crispo-multi-aspect-critique-suggestion","title":"CriSPO: Multi-Aspect Critique-Suggestion-guided Automatic Prompt Optimization for Text Generation","date":"2024-10-03","arxiv_id":"2410.02748","repositories_listed":1,"syntology":null},{"url":"/paper/salient-information-prompting-to-steer","slug":"salient-information-prompting-to-steer","title":"Salient Information Prompting to Steer Content in Prompt-based Abstractive Summarization","date":"2024-10-03","arxiv_id":"2410.02741","repositories_listed":1,"syntology":null},{"url":"/paper/bordirlines-a-dataset-for-evaluating-cross","slug":"bordirlines-a-dataset-for-evaluating-cross","title":"BordIRlines: A Dataset for Evaluating Cross-lingual Retrieval-Augmented Generation","date":"2024-10-02","arxiv_id":"2410.01171","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/bordirlines-a-dataset-for-evaluating-cross#ran","syntology_url":"https://syntology.ai/paper/2410.01171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01171"}},"official":{"repos":["manestay/bordirlines"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/factalign-long-form-factuality-alignment-of","slug":"factalign-long-form-factuality-alignment-of","title":"FactAlign: Long-form Factuality Alignment of Large Language Models","date":"2024-10-02","arxiv_id":"2410.01691","repositories_listed":1,"syntology":null},{"url":"/paper/scvlm-a-vision-language-model-for-driving","slug":"scvlm-a-vision-language-model-for-driving","title":"ScVLM: Enhancing Vision-Language Model for Safety-Critical Event Understanding","date":"2024-10-01","arxiv_id":"2410.00982","repositories_listed":1,"syntology":null},{"url":"/paper/faitheval-can-your-language-model-stay","slug":"faitheval-can-your-language-model-stay","title":"FaithEval: Can Your Language Model Stay Faithful to Context, Even If \"The Moon is Made of Marshmallows\"","date":"2024-09-30","arxiv_id":"2410.03727","repositories_listed":1,"syntology":null},{"url":"/paper/helpd-mitigating-hallucination-of-lvlms-by","slug":"helpd-mitigating-hallucination-of-lvlms-by","title":"HELPD: Mitigating Hallucination of LVLMs by Hierarchical Feedback Learning with Vision-enhanced Penalty Decoding","date":"2024-09-30","arxiv_id":"2409.20429","repositories_listed":1,"syntology":null},{"url":"/paper/llm-hallucinations-in-practical-code","slug":"llm-hallucinations-in-practical-code","title":"LLM Hallucinations in Practical Code Generation: Phenomena, Mechanism, and Mitigation","date":"2024-09-30","arxiv_id":"2409.20550","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/llm-hallucinations-in-practical-code#ran","syntology_url":"https://syntology.ai/paper/2409.20550","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20550"}},"official":{"repos":["deepsoftwareanalytics/llmcodinghallucination"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/eventhallusion-diagnosing-event","slug":"eventhallusion-diagnosing-event","title":"EventHallusion: Diagnosing Event Hallucinations in Video LLMs","date":"2024-09-25","arxiv_id":"2409.16597","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eventhallusion-diagnosing-event#ran","syntology_url":"https://syntology.ai/paper/2409.16597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16597"}},"official":{"repos":["stevetich/eventhallusion"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/pre-trained-language-models-return","slug":"pre-trained-language-models-return","title":"Pre-trained Language Models Return Distinguishable Probability Distributions to Unfaithfully Hallucinated Texts","date":"2024-09-25","arxiv_id":"2409.16658","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-hallucination-mitigation-framework","slug":"a-unified-hallucination-mitigation-framework","title":"A Unified Hallucination Mitigation Framework for Large Vision-Language Models","date":"2024-09-24","arxiv_id":"2409.16494","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-risk-of-retrieval-augmented","slug":"controlling-risk-of-retrieval-augmented","title":"Controlling Risk of Retrieval-augmented Generation: A Counterfactual Prompting Framework","date":"2024-09-24","arxiv_id":"2409.16146","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/controlling-risk-of-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2409.16146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16146"}},"official":{"repos":["ict-bigdatalab/rc-rag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/xtrust-on-the-multilingual-trustworthiness-of","slug":"xtrust-on-the-multilingual-trustworthiness-of","title":"XTRUST: On the Multilingual Trustworthiness of Large Language Models","date":"2024-09-24","arxiv_id":"2409.15762","repositories_listed":1,"syntology":null},{"url":"/paper/parse-trees-guided-llm-prompt-compression","slug":"parse-trees-guided-llm-prompt-compression","title":"Parse Trees Guided LLM Prompt Compression","date":"2024-09-23","arxiv_id":"2409.15395","repositories_listed":1,"syntology":null},{"url":"/paper/fair-gpt-a-virtual-consultant-for-research","slug":"fair-gpt-a-virtual-consultant-for-research","title":"FAIR GPT: A virtual consultant for research data management in ChatGPT","date":"2024-09-20","arxiv_id":"2410.07108","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-image-hallucination-in-text-to","slug":"evaluating-image-hallucination-in-text-to","title":"Evaluating Image Hallucination in Text-to-Image Generation with Question-Answering","date":"2024-09-19","arxiv_id":"2409.12784","repositories_listed":1,"syntology":null},{"url":"/paper/journeybench-a-challenging-one-stop-vision","slug":"journeybench-a-challenging-one-stop-vision","title":"JourneyBench: A Challenging One-Stop Vision-Language Understanding Benchmark of Generated Images","date":"2024-09-19","arxiv_id":"2409.12953","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/journeybench-a-challenging-one-stop-vision#ran","syntology_url":"https://syntology.ai/paper/2409.12953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12953"}},"official":{"repos":["journeybench/journeybench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-comprehensive-evaluation-of-quantized","slug":"a-comprehensive-evaluation-of-quantized","title":"Exploring the Trade-Offs: Quantization Methods, Task Difficulty, and Model Size in Large Language Models From Edge to Giant","date":"2024-09-17","arxiv_id":"2409.11055","repositories_listed":1,"syntology":null},{"url":"/paper/thames-an-end-to-end-tool-for-hallucination","slug":"thames-an-end-to-end-tool-for-hallucination","title":"THaMES: An End-to-End Tool for Hallucination Mitigation and Evaluation in Large Language Models","date":"2024-09-17","arxiv_id":"2409.11353","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/thames-an-end-to-end-tool-for-hallucination#ran","syntology_url":"https://syntology.ai/paper/2409.11353","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.11353"}},"official":{"repos":["holistic-ai/THaMES"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/halo-hallucination-analysis-and-learning","slug":"halo-hallucination-analysis-and-learning","title":"HALO: Hallucination Analysis and Learning Optimization to Empower LLMs with Retrieval-Augmented Context for Guided Clinical Decision Making","date":"2024-09-16","arxiv_id":"2409.10011","repositories_listed":1,"syntology":null},{"url":"/paper/trustworthiness-in-retrieval-augmented","slug":"trustworthiness-in-retrieval-augmented","title":"Trustworthiness in Retrieval-Augmented Generation Systems: A Survey","date":"2024-09-16","arxiv_id":"2409.10102","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/trustworthiness-in-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2409.10102","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.10102"}},"official":{"repos":["smallporridge/trustworthyrag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/confidence-estimation-for-llm-based-dialogue","slug":"confidence-estimation-for-llm-based-dialogue","title":"Confidence Estimation for LLM-Based Dialogue State Tracking","date":"2024-09-15","arxiv_id":"2409.09629","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/confidence-estimation-for-llm-based-dialogue#ran","syntology_url":"https://syntology.ai/paper/2409.09629","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.09629"}},"official":{"repos":["jennycs0830/confidence_score_dst"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/generating-faithful-and-salient-text-from","slug":"generating-faithful-and-salient-text-from","title":"Generating Faithful and Salient Text from Multimodal Data","date":"2024-09-06","arxiv_id":"2409.03961","repositories_listed":1,"syntology":null},{"url":"/paper/hallucination-detection-in-llms-fast-and","slug":"hallucination-detection-in-llms-fast-and","title":"Hallucination Detection in LLMs: Fast and Memory-Efficient Fine-Tuned Models","date":"2024-09-04","arxiv_id":"2409.02976","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-multimodal-hallucination-with","slug":"understanding-multimodal-hallucination-with","title":"Understanding Multimodal Hallucination with Parameter-Free Representation Alignment","date":"2024-09-02","arxiv_id":"2409.01151","repositories_listed":1,"syntology":null},{"url":"/paper/look-compare-decide-alleviating-hallucination","slug":"look-compare-decide-alleviating-hallucination","title":"Look, Compare, Decide: Alleviating Hallucination in Large Vision-Language Models via Multi-View Multi-Path Reasoning","date":"2024-08-30","arxiv_id":"2408.17150","repositories_listed":1,"syntology":null},{"url":"/paper/towards-empathetic-conversational-recommender","slug":"towards-empathetic-conversational-recommender","title":"Towards Empathetic Conversational Recommender Systems","date":"2024-08-30","arxiv_id":"2409.10527","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/towards-empathetic-conversational-recommender#ran","syntology_url":"https://syntology.ai/paper/2409.10527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.10527"}},"official":{"repos":["zxd-octopus/ECR"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/llava-mod-making-llava-tiny-via-moe-knowledge","slug":"llava-mod-making-llava-tiny-via-moe-knowledge","title":"LLaVA-MoD: Making LLaVA Tiny via MoE Knowledge Distillation","date":"2024-08-28","arxiv_id":"2408.15881","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/llava-mod-making-llava-tiny-via-moe-knowledge#ran","syntology_url":"https://syntology.ai/paper/2408.15881","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.15881"}},"official":{"repos":["shufangxun/llava-mod"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/vlm4bio-a-benchmark-dataset-to-evaluate","slug":"vlm4bio-a-benchmark-dataset-to-evaluate","title":"VLM4Bio: A Benchmark Dataset to Evaluate Pretrained Vision-Language Models for Trait Discovery from Biological Images","date":"2024-08-28","arxiv_id":"2408.16176","repositories_listed":1,"syntology":null},{"url":"/paper/convis-contrastive-decoding-with","slug":"convis-contrastive-decoding-with","title":"ConVis: Contrastive Decoding with Hallucination Visualization for Mitigating Hallucinations in Multimodal Large Language Models","date":"2024-08-25","arxiv_id":"2408.13906","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/convis-contrastive-decoding-with#ran","syntology_url":"https://syntology.ai/paper/2408.13906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.13906"}},"official":{"repos":["yejipark-m/convis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/genetic-approach-to-mitigate-hallucination-in","slug":"genetic-approach-to-mitigate-hallucination-in","title":"Genetic Approach to Mitigate Hallucination in Generative IR","date":"2024-08-25","arxiv_id":"2409.00085","repositories_listed":1,"syntology":null},{"url":"/paper/graph-retrieval-augmented-trustworthiness","slug":"graph-retrieval-augmented-trustworthiness","title":"GRATR: Zero-Shot Evidence Graph Retrieval-Augmented Trustworthiness Reasoning","date":"2024-08-22","arxiv_id":"2408.12333","repositories_listed":1,"syntology":null},{"url":"/paper/improving-factuality-in-large-language-models","slug":"improving-factuality-in-large-language-models","title":"Improving Factuality in Large Language Models via Decoding-Time Hallucinatory and Truthful Comparators","date":"2024-08-22","arxiv_id":"2408.12325","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":4,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-factuality-in-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2408.12325","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12325"}},"official":{"repos":["ydk122024/cdt"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rovrm-a-robust-visual-reward-model-optimized","slug":"rovrm-a-robust-visual-reward-model-optimized","title":"RoVRM: A Robust Visual Reward Model Optimized via Auxiliary Textual Preference Data","date":"2024-08-22","arxiv_id":"2408.12109","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/rovrm-a-robust-visual-reward-model-optimized#ran","syntology_url":"https://syntology.ai/paper/2408.12109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.12109"}},"official":{"repos":["wangclnlp/vision-llm-alignment"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/slm-meets-llm-balancing-latency","slug":"slm-meets-llm-balancing-latency","title":"SLM Meets LLM: Balancing Latency, Interpretability and Consistency in Hallucination Detection","date":"2024-08-22","arxiv_id":"2408.12748","repositories_listed":1,"syntology":null},{"url":"/paper/reefknot-a-comprehensive-benchmark-for","slug":"reefknot-a-comprehensive-benchmark-for","title":"Reefknot: A Comprehensive Benchmark for Relation Hallucination Evaluation, Analysis and Mitigation in Multimodal Large Language Models","date":"2024-08-18","arxiv_id":"2408.09429","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reefknot-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2408.09429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09429"}},"official":{"repos":["JackChen-seu/Reefknot"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/graph-retrieval-augmented-generation-a-survey","slug":"graph-retrieval-augmented-generation-a-survey","title":"Graph Retrieval-Augmented Generation: A Survey","date":"2024-08-15","arxiv_id":"2408.08921","repositories_listed":1,"syntology":null},{"url":"/paper/ssl-a-self-similarity-loss-for-improving","slug":"ssl-a-self-similarity-loss-for-improving","title":"SSL: A Self-similarity Loss for Improving Generative Image Super-resolution","date":"2024-08-11","arxiv_id":"2408.05713","repositories_listed":1,"syntology":null},{"url":"/paper/order-matters-in-hallucination-reasoning","slug":"order-matters-in-hallucination-reasoning","title":"Order Matters in Hallucination: Reasoning Order as Benchmark and Reflexive Prompting for Large-Language-Models","date":"2024-08-09","arxiv_id":"2408.05093","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/order-matters-in-hallucination-reasoning#ran","syntology_url":"https://syntology.ai/paper/2408.05093","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.05093"}},"official":{"repos":["xiezikai/reflexiveprompting"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/handwritten-code-recognition-for-pen-and","slug":"handwritten-code-recognition-for-pen-and","title":"Handwritten Code Recognition for Pen-and-Paper CS Education","date":"2024-08-07","arxiv_id":"2408.07220","repositories_listed":1,"syntology":null},{"url":"/paper/2408-02032","slug":"2408-02032","title":"Self-Introspective Decoding: Alleviating Hallucinations for Large Vision-Language Models","date":"2024-08-04","arxiv_id":"2408.02032","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-02032#ran","syntology_url":"https://syntology.ai/paper/2408.02032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.02032"}},"official":{"repos":["huofushuo/SID"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/2408-01262","slug":"2408-01262","title":"RAGEval: Scenario Specific RAG Evaluation Dataset Generation Framework","date":"2024-08-02","arxiv_id":"2408.01262","repositories_listed":1,"syntology":null},{"url":"/paper/2408-01355","slug":"2408-01355","title":"Hallu-PI: Evaluating Hallucination in Multi-modal Large Language Models within Perturbed Inputs","date":"2024-08-02","arxiv_id":"2408.01355","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00357","slug":"2408-00357","title":"DeliLaw: A Chinese Legal Counselling System Based on a Large Language Model","date":"2024-08-01","arxiv_id":"2408.00357","repositories_listed":1,"syntology":null},{"url":"/paper/2408-00550","slug":"2408-00550","title":"Mitigating Multilingual Hallucination in Large Vision-Language Models","date":"2024-08-01","arxiv_id":"2408.00550","repositories_listed":1,"syntology":null},{"url":"/paper/paying-more-attention-to-image-a-training","slug":"paying-more-attention-to-image-a-training","title":"Paying More Attention to Image: A Training-Free Method for Alleviating Hallucination in LVLMs","date":"2024-07-31","arxiv_id":"2407.21771","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/paying-more-attention-to-image-a-training#ran","syntology_url":"https://syntology.ai/paper/2407.21771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.21771"}},"official":null}},{"url":"/paper/automated-review-generation-method-based-on","slug":"automated-review-generation-method-based-on","title":"Automated Review Generation Method Based on Large Language Models","date":"2024-07-30","arxiv_id":"2407.20906","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/automated-review-generation-method-based-on#ran","syntology_url":"https://syntology.ai/paper/2407.20906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.20906"}},"official":{"repos":["tju-ecat-ai/automaticreviewgeneration"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-llm-s-cognition-via-structurization","slug":"enhancing-llm-s-cognition-via-structurization","title":"Enhancing LLM's Cognition via Structurization","date":"2024-07-23","arxiv_id":"2407.16434","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-llm-s-cognition-via-structurization#ran","syntology_url":"https://syntology.ai/paper/2407.16434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16434"}},"official":{"repos":["alibaba/struxgpt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/machine-translation-hallucination-detection","slug":"machine-translation-hallucination-detection","title":"Machine Translation Hallucination Detection for Low and High Resource Languages using Large Language Models","date":"2024-07-23","arxiv_id":"2407.16470","repositories_listed":1,"syntology":null},{"url":"/paper/retrieve-generate-evaluate-a-case-study-for","slug":"retrieve-generate-evaluate-a-case-study-for","title":"Retrieve, Generate, Evaluate: A Case Study for Medical Paraphrases Generation with Small Language Models","date":"2024-07-23","arxiv_id":"2407.16565","repositories_listed":1,"syntology":null},{"url":"/paper/haloquest-a-visual-hallucination-dataset-for","slug":"haloquest-a-visual-hallucination-dataset-for","title":"HaloQuest: A Visual Hallucination Dataset for Advancing Multimodal Reasoning","date":"2024-07-22","arxiv_id":"2407.15680","repositories_listed":1,"syntology":null},{"url":"/paper/data-centric-human-preference-optimization","slug":"data-centric-human-preference-optimization","title":"Data-Centric Human Preference Optimization with Rationales","date":"2024-07-19","arxiv_id":"2407.14477","repositories_listed":1,"syntology":null},{"url":"/paper/anhalten-cross-lingual-transfer-for-german","slug":"anhalten-cross-lingual-transfer-for-german","title":"ANHALTEN: Cross-Lingual Transfer for German Token-Level Reference-Free Hallucination Detection","date":"2024-07-18","arxiv_id":"2407.13702","repositories_listed":1,"syntology":null},{"url":"/paper/halu-j-critique-based-hallucination-judge","slug":"halu-j-critique-based-hallucination-judge","title":"Halu-J: Critique-Based Hallucination Judge","date":"2024-07-17","arxiv_id":"2407.12943","repositories_listed":1,"syntology":null},{"url":"/paper/fine-grained-hallucination-detection-and-1","slug":"fine-grained-hallucination-detection-and-1","title":"Localizing and Mitigating Errors in Long-form Question Answering","date":"2024-07-16","arxiv_id":"2407.11930","repositories_listed":1,"syntology":null},{"url":"/paper/what-s-wrong-refining-meeting-summaries-with","slug":"what-s-wrong-refining-meeting-summaries-with","title":"What's Wrong? Refining Meeting Summaries with LLM Feedback","date":"2024-07-16","arxiv_id":"2407.11919","repositories_listed":1,"syntology":null},{"url":"/paper/learning-dynamics-of-llm-finetuning","slug":"learning-dynamics-of-llm-finetuning","title":"Learning Dynamics of LLM Finetuning","date":"2024-07-15","arxiv_id":"2407.10490","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-dynamics-of-llm-finetuning#ran","syntology_url":"https://syntology.ai/paper/2407.10490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10490"}},"official":{"repos":["joshua-ren/learning_dynamics_llm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/synergistic-multi-agent-framework-with","slug":"synergistic-multi-agent-framework-with","title":"Synergistic Multi-Agent Framework with Trajectory Learning for Knowledge-Intensive Tasks","date":"2024-07-13","arxiv_id":"2407.09893","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-entity-level-hallucination-in","slug":"mitigating-entity-level-hallucination-in","title":"Mitigating Entity-Level Hallucination in Large Language Models","date":"2024-07-12","arxiv_id":"2407.09417","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-universal-truthfulness-hyperplane","slug":"on-the-universal-truthfulness-hyperplane","title":"On the Universal Truthfulness Hyperplane Inside LLMs","date":"2024-07-11","arxiv_id":"2407.08582","repositories_listed":1,"syntology":null},{"url":"/paper/lookback-lens-detecting-and-mitigating","slug":"lookback-lens-detecting-and-mitigating","title":"Lookback Lens: Detecting and Mitigating Contextual Hallucinations in Large Language Models Using Only Attention Maps","date":"2024-07-09","arxiv_id":"2407.07071","repositories_listed":1,"syntology":{"n":20,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":20,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/lookback-lens-detecting-and-mitigating#ran","syntology_url":"https://syntology.ai/paper/2407.07071","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07071"}},"official":{"repos":["voidism/lookback-lens"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/kg-fpq-evaluating-factuality-hallucination-in","slug":"kg-fpq-evaluating-factuality-hallucination-in","title":"KG-FPQ: Evaluating Factuality Hallucination in LLMs with Knowledge Graph-based False Premise Questions","date":"2024-07-08","arxiv_id":"2407.05868","repositories_listed":1,"syntology":null},{"url":"/paper/llm-based-open-domain-integrated-task-and","slug":"llm-based-open-domain-integrated-task-and","title":"Controllable and Reliable Knowledge-Intensive Task-Oriented Conversational Agents with Declarative Genie Worksheets","date":"2024-07-08","arxiv_id":"2407.05674","repositories_listed":1,"syntology":null},{"url":"/paper/multi-object-hallucination-in-vision-language","slug":"multi-object-hallucination-in-vision-language","title":"Multi-Object Hallucination in Vision-Language Models","date":"2024-07-08","arxiv_id":"2407.06192","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-object-hallucination-in-vision-language#ran","syntology_url":"https://syntology.ai/paper/2407.06192","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06192"}},"official":{"repos":["sled-group/moh"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-hallucination-detection-through","slug":"enhancing-hallucination-detection-through","title":"Enhancing Hallucination Detection through Perturbation-Based Synthetic Data Generation in System Responses","date":"2024-07-07","arxiv_id":"2407.05474","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-hallucination-detection-through#ran","syntology_url":"https://syntology.ai/paper/2407.05474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.05474"}},"official":{"repos":["asappresearch/halugen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/anah-v2-scaling-analytical-hallucination","slug":"anah-v2-scaling-analytical-hallucination","title":"ANAH-v2: Scaling Analytical Hallucination Annotation of Large Language Models","date":"2024-07-05","arxiv_id":"2407.04693","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/anah-v2-scaling-analytical-hallucination#ran","syntology_url":"https://syntology.ai/paper/2407.04693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04693"}},"official":{"repos":["open-compass/anah"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mj-bench-is-your-multimodal-reward-model","slug":"mj-bench-is-your-multimodal-reward-model","title":"MJ-Bench: Is Your Multimodal Reward Model Really a Good Judge for Text-to-Image Generation?","date":"2024-07-05","arxiv_id":"2407.04842","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/mj-bench-is-your-multimodal-reward-model#ran","syntology_url":"https://syntology.ai/paper/2407.04842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04842"}},"official":{"repos":["MJ-Bench/MJ-Bench"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-internal-states-reveal-hallucination-risk","slug":"llm-internal-states-reveal-hallucination-risk","title":"LLM Internal States Reveal Hallucination Risk Faced With a Query","date":"2024-07-03","arxiv_id":"2407.03282","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm-internal-states-reveal-hallucination-risk#ran","syntology_url":"https://syntology.ai/paper/2407.03282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03282"}},"official":{"repos":["ziweiji/Internal_States_Reveal_Hallucination"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/medvh-towards-systematic-evaluation-of","slug":"medvh-towards-systematic-evaluation-of","title":"MedVH: Towards Systematic Evaluation of Hallucination for Large Vision Language Models in the Medical Context","date":"2024-07-03","arxiv_id":"2407.02730","repositories_listed":1,"syntology":null},{"url":"/paper/mememo-on-device-retrieval-augmentation-for","slug":"mememo-on-device-retrieval-augmentation-for","title":"MeMemo: On-device Retrieval Augmentation for Private and Personalized Text Generation","date":"2024-07-02","arxiv_id":"2407.01972","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-involuntary-truth","slug":"large-language-models-are-involuntary-truth","title":"Large Language Models Are Involuntary Truth-Tellers: Exploiting Fallacy Failure for Jailbreak Attacks","date":"2024-07-01","arxiv_id":"2407.00869","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/large-language-models-are-involuntary-truth#ran","syntology_url":"https://syntology.ai/paper/2407.00869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00869"}},"official":{"repos":["Yue-LLM-Pit/FFA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-and-mitigating-the-multimodal","slug":"investigating-and-mitigating-the-multimodal","title":"Investigating and Mitigating the Multimodal Hallucination Snowballing in Large Vision-Language Models","date":"2024-06-30","arxiv_id":"2407.00569","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/investigating-and-mitigating-the-multimodal#ran","syntology_url":"https://syntology.ai/paper/2407.00569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00569"}},"official":{"repos":["whongzhong/MMHalSnowball"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/biokgbench-a-knowledge-graph-checking","slug":"biokgbench-a-knowledge-graph-checking","title":"BioKGBench: A Knowledge Graph Checking Benchmark of AI Agent for Biomedical Science","date":"2024-06-29","arxiv_id":"2407.00466","repositories_listed":1,"syntology":null},{"url":"/paper/grapharena-benchmarking-large-language-models","slug":"grapharena-benchmarking-large-language-models","title":"GraphArena: Benchmarking Large Language Models on Graph Computational Problems","date":"2024-06-29","arxiv_id":"2407.00379","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/grapharena-benchmarking-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2407.00379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00379"}},"official":{"repos":["squareroot3/grapharena"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/toolbehonest-a-multi-level-hallucination","slug":"toolbehonest-a-multi-level-hallucination","title":"ToolBeHonest: A Multi-level Hallucination Diagnostic Benchmark for Tool-Augmented Large Language Models","date":"2024-06-28","arxiv_id":"2406.20015","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/toolbehonest-a-multi-level-hallucination#ran","syntology_url":"https://syntology.ai/paper/2406.20015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20015"}},"official":{"repos":["toolbehonest/toolbehonest"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/from-artificial-needles-to-real-haystacks","slug":"from-artificial-needles-to-real-haystacks","title":"From Artificial Needles to Real Haystacks: Improving Retrieval Capabilities in LLMs by Finetuning on Synthetic Data","date":"2024-06-27","arxiv_id":"2406.19292","repositories_listed":1,"syntology":null},{"url":"/paper/handling-ontology-gaps-in-semantic-parsing","slug":"handling-ontology-gaps-in-semantic-parsing","title":"Handling Ontology Gaps in Semantic Parsing","date":"2024-06-27","arxiv_id":"2406.19537","repositories_listed":1,"syntology":null},{"url":"/paper/understand-what-llm-needs-dual-preference","slug":"understand-what-llm-needs-dual-preference","title":"Understand What LLM Needs: Dual Preference Alignment for Retrieval-Augmented Generation","date":"2024-06-26","arxiv_id":"2406.18676","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/understand-what-llm-needs-dual-preference#ran","syntology_url":"https://syntology.ai/paper/2406.18676","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18676"}},"official":{"repos":["dongguanting/dpa-rag"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-hallucination-in-fictional","slug":"mitigating-hallucination-in-fictional","title":"Mitigating Hallucination in Fictional Character Role-Play","date":"2024-06-25","arxiv_id":"2406.17260","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-and-analyzing-relationship","slug":"evaluating-and-analyzing-relationship","title":"Evaluating and Analyzing Relationship Hallucinations in Large Vision-Language Models","date":"2024-06-24","arxiv_id":"2406.16449","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/evaluating-and-analyzing-relationship#ran","syntology_url":"https://syntology.ai/paper/2406.16449","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16449"}},"official":{"repos":["mrwu-mac/R-Bench"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-the-quality-of-hallucination","slug":"evaluating-the-quality-of-hallucination","title":"Evaluating the Quality of Hallucination Benchmarks for Large Vision-Language Models","date":"2024-06-24","arxiv_id":"2406.17115","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-entropy-probes-robust-and-cheap","slug":"semantic-entropy-probes-robust-and-cheap","title":"Semantic Entropy Probes: Robust and Cheap Hallucination Detection in LLMs","date":"2024-06-22","arxiv_id":"2406.15927","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/semantic-entropy-probes-robust-and-cheap#ran","syntology_url":"https://syntology.ai/paper/2406.15927","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.15927"}},"official":null}},{"url":"/paper/evaluating-rag-fusion-with-ragelo-an","slug":"evaluating-rag-fusion-with-ragelo-an","title":"Evaluating RAG-Fusion with RAGElo: an Automated Elo-based Framework","date":"2024-06-20","arxiv_id":"2406.14783","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evaluating-rag-fusion-with-ragelo-an#ran","syntology_url":"https://syntology.ai/paper/2406.14783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14783"}},"official":{"repos":["zetaalphavector/ragelo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-graph-enhanced-large-language","slug":"knowledge-graph-enhanced-large-language","title":"Knowledge Graph-Enhanced Large Language Models via Path Selection","date":"2024-06-19","arxiv_id":"2406.13862","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowledge-graph-enhanced-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.13862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13862"}},"official":{"repos":["haochenliu2000/kelp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rethinking-abdominal-organ-segmentation-raos","slug":"rethinking-abdominal-organ-segmentation-raos","title":"Rethinking Abdominal Organ Segmentation (RAOS) in the clinical scenario: A robustness evaluation benchmark with challenging cases","date":"2024-06-19","arxiv_id":"2406.13674","repositories_listed":1,"syntology":null},{"url":"/paper/stackrag-agent-improving-developer-answers","slug":"stackrag-agent-improving-developer-answers","title":"StackRAG Agent: Improving Developer Answers with Retrieval-Augmented Generation","date":"2024-06-19","arxiv_id":"2406.13840","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-errors-through-ensembling-prompts","slug":"detecting-errors-through-ensembling-prompts","title":"Detecting Errors through Ensembling Prompts (DEEP): An End-to-End LLM Framework for Detecting Factual Errors","date":"2024-06-18","arxiv_id":"2406.13009","repositories_listed":1,"syntology":null},{"url":"/paper/fast-and-slow-generating-an-empirical-study","slug":"fast-and-slow-generating-an-empirical-study","title":"Fast and Slow Generating: An Empirical Study on Large and Small Language Models Collaborative Decoding","date":"2024-06-18","arxiv_id":"2406.12295","repositories_listed":1,"syntology":null},{"url":"/paper/on-policy-fine-grained-knowledge-feedback-for","slug":"on-policy-fine-grained-knowledge-feedback-for","title":"On-Policy Fine-grained Knowledge Feedback for Hallucination Mitigation","date":"2024-06-18","arxiv_id":"2406.12221","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-policy-fine-grained-knowledge-feedback-for#ran","syntology_url":"https://syntology.ai/paper/2406.12221","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12221"}},"official":null}},{"url":"/paper/counterfactual-debating-with-preset-stances","slug":"counterfactual-debating-with-preset-stances","title":"Counterfactual Debating with Preset Stances for Hallucination Elimination of LLMs","date":"2024-06-17","arxiv_id":"2406.11514","repositories_listed":1,"syntology":null},{"url":"/paper/mdpo-conditional-preference-optimization-for","slug":"mdpo-conditional-preference-optimization-for","title":"mDPO: Conditional Preference Optimization for Multimodal Large Language Models","date":"2024-06-17","arxiv_id":"2406.11839","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mdpo-conditional-preference-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2406.11839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11839"}},"official":{"repos":["luka-group/mDPO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-needle-in-a-haystack-benchmarking","slug":"multimodal-needle-in-a-haystack-benchmarking","title":"Multimodal Needle in a Haystack: Benchmarking Long-Context Capability of Multimodal Large Language Models","date":"2024-06-17","arxiv_id":"2406.11230","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multimodal-needle-in-a-haystack-benchmarking#ran","syntology_url":"https://syntology.ai/paper/2406.11230","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11230"}},"official":{"repos":["wang-ml-lab/multimodal-needle-in-a-haystack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-training-large-language-models-through","slug":"self-training-large-language-models-through","title":"Self-training Large Language Models through Knowledge Detection","date":"2024-06-17","arxiv_id":"2406.11275","repositories_listed":1,"syntology":null},{"url":"/paper/small-agent-can-also-rock-empowering-small","slug":"small-agent-can-also-rock-empowering-small","title":"Small Agent Can Also Rock! Empowering Small Language Models as Hallucination Detector","date":"2024-06-17","arxiv_id":"2406.11277","repositories_listed":1,"syntology":null},{"url":"/paper/texttt-moe-rbench-towards-building-reliable","slug":"texttt-moe-rbench-towards-building-reliable","title":"$\\texttt{MoE-RBench}$: Towards Building Reliable Language Models with Sparse Mixture-of-Experts","date":"2024-06-17","arxiv_id":"2406.11353","repositories_listed":1,"syntology":null},{"url":"/paper/post-hoc-utterance-refining-method-by-entity","slug":"post-hoc-utterance-refining-method-by-entity","title":"Post-hoc Utterance Refining Method by Entity Mining for Faithful Knowledge Grounded Conversations","date":"2024-06-16","arxiv_id":"2406.10809","repositories_listed":1,"syntology":null},{"url":"/paper/defan-definitive-answer-dataset-for-llms","slug":"defan-definitive-answer-dataset-for-llms","title":"DefAn: Definitive Answer Dataset for LLMs Hallucination Evaluation","date":"2024-06-13","arxiv_id":"2406.09155","repositories_listed":1,"syntology":null},{"url":"/paper/mmrel-a-relation-understanding-dataset-and","slug":"mmrel-a-relation-understanding-dataset-and","title":"MMRel: A Relation Understanding Benchmark in the MLLM Era","date":"2024-06-13","arxiv_id":"2406.09121","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-hallucinations-in-diffusion","slug":"understanding-hallucinations-in-diffusion","title":"Understanding Hallucinations in Diffusion Models through Mode Interpolation","date":"2024-06-13","arxiv_id":"2406.09358","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/understanding-hallucinations-in-diffusion#ran","syntology_url":"https://syntology.ai/paper/2406.09358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09358"}},"official":{"repos":["locuslab/diffusion-model-hallucination"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/we-have-a-package-for-you-a-comprehensive","slug":"we-have-a-package-for-you-a-comprehensive","title":"We Have a Package for You! A Comprehensive Analysis of Package Hallucinations by Code Generating LLMs","date":"2024-06-12","arxiv_id":"2406.10279","repositories_listed":1,"syntology":null}],"record_sha256":"42b4fbbb1d9604651928506849938bfddfdefe61cfc020877d89314c6d2a0762","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}