{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/hallucination/papers/5","list_of":"/task/hallucination","task":"Hallucination","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":19,"rows_per_page":100,"rows":[401,500],"of":1816,"counts":{"archive_papers_tagged":1816,"with_a_code_link":752,"where_syntology_ran_a_sample":276,"not_listed_spam_title":0,"listed":1816,"listed_where_code_ran":276,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":240,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":240,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/hallucination","prev":"/task/hallucination/papers/4","next":"/task/hallucination/papers/6","papers":[{"url":"/paper/a-probabilistic-framework-for-llm","slug":"a-probabilistic-framework-for-llm","title":"A Probabilistic Framework for LLM Hallucination Detection via Belief Tree Propagation","date":"2024-06-11","arxiv_id":"2406.06950","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-probabilistic-framework-for-llm#ran","syntology_url":"https://syntology.ai/paper/2406.06950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06950"}},"official":{"repos":["ucsb-nlp-chang/btprop"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/image-textualization-an-automatic-framework","slug":"image-textualization-an-automatic-framework","title":"Image Textualization: An Automatic Framework for Creating Accurate and Detailed Image Descriptions","date":"2024-06-11","arxiv_id":"2406.07502","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/image-textualization-an-automatic-framework#ran","syntology_url":"https://syntology.ai/paper/2406.07502","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07502"}},"official":{"repos":["sterzhang/image-textualization"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-hallucination-in-simultaneous-machine","slug":"on-the-hallucination-in-simultaneous-machine","title":"On the Hallucination in Simultaneous Machine Translation","date":"2024-06-11","arxiv_id":"2406.07239","repositories_listed":1,"syntology":null},{"url":"/paper/real-sampling-boosting-factuality-and","slug":"real-sampling-boosting-factuality-and","title":"REAL Sampling: Boosting Factuality and Diversity of Open-Ended Generation via Asymptotic Entropy","date":"2024-06-11","arxiv_id":"2406.07735","repositories_listed":1,"syntology":null},{"url":"/paper/3d-grand-towards-better-grounding-and-less","slug":"3d-grand-towards-better-grounding-and-less","title":"3D-GRAND: A Million-Scale Dataset for 3D-LLMs with Better Grounding and Less Hallucination","date":"2024-06-07","arxiv_id":"2406.05132","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3d-grand-towards-better-grounding-and-less#ran","syntology_url":"https://syntology.ai/paper/2406.05132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05132"}},"official":{"repos":["sled-group/3D-GRAND"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-on-parameter-efficient","slug":"an-empirical-study-on-parameter-efficient","title":"An Empirical Study on Parameter-Efficient Fine-Tuning for MultiModal Large Language Models","date":"2024-06-07","arxiv_id":"2406.05130","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-on-parameter-efficient#ran","syntology_url":"https://syntology.ai/paper/2406.05130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05130"}},"official":{"repos":["alenai97/peft-mllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ottawa-optimal-transport-adaptive-word","slug":"ottawa-optimal-transport-adaptive-word","title":"OTTAWA: Optimal TransporT Adaptive Word Aligner for Hallucination and Omission Translation Errors Detection","date":"2024-06-04","arxiv_id":"2406.01919","repositories_listed":1,"syntology":null},{"url":"/paper/dafnet-dynamic-auxiliary-fusion-for","slug":"dafnet-dynamic-auxiliary-fusion-for","title":"DAFNet: Dynamic Auxiliary Fusion for Sequential Model Editing in Large Language Models","date":"2024-05-31","arxiv_id":"2405.20588","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-noise-robustness-of-retrieval","slug":"enhancing-noise-robustness-of-retrieval","title":"Enhancing Noise Robustness of Retrieval-Augmented Language Models with Adaptive Adversarial Training","date":"2024-05-31","arxiv_id":"2405.20978","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-noise-robustness-of-retrieval#ran","syntology_url":"https://syntology.ai/paper/2405.20978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20978"}},"official":{"repos":["calubkk/raat"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/anah-analytical-annotation-of-hallucinations","slug":"anah-analytical-annotation-of-hallucinations","title":"ANAH: Analytical Annotation of Hallucinations in Large Language Models","date":"2024-05-30","arxiv_id":"2405.20315","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/anah-analytical-annotation-of-hallucinations#ran","syntology_url":"https://syntology.ai/paper/2405.20315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20315"}},"official":{"repos":["open-compass/anah"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/noiseboost-alleviating-hallucination-with","slug":"noiseboost-alleviating-hallucination-with","title":"NoiseBoost: Alleviating Hallucination with Noise Perturbation for Multimodal Large Language Models","date":"2024-05-30","arxiv_id":"2405.20081","repositories_listed":1,"syntology":null},{"url":"/paper/llms-and-memorization-on-quality-and","slug":"llms-and-memorization-on-quality-and","title":"LLMs and Memorization: On Quality and Specificity of Copyright Compliance","date":"2024-05-28","arxiv_id":"2405.18492","repositories_listed":1,"syntology":null},{"url":"/paper/personalized-steering-of-large-language","slug":"personalized-steering-of-large-language","title":"Personalized Steering of Large Language Models: Versatile Steering Vectors Through Bi-directional Preference Optimization","date":"2024-05-28","arxiv_id":"2406.00045","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/personalized-steering-of-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.00045","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00045"}},"official":{"repos":["CaoYuanpu/BiPO"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/timechara-evaluating-point-in-time-character","slug":"timechara-evaluating-point-in-time-character","title":"TimeChara: Evaluating Point-in-Time Character Hallucination of Role-Playing Large Language Models","date":"2024-05-28","arxiv_id":"2405.18027","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/timechara-evaluating-point-in-time-character#ran","syntology_url":"https://syntology.ai/paper/2405.18027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18027"}},"official":{"repos":["ahnjaewoo/timechara"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/think-before-you-act-a-two-stage-framework","slug":"think-before-you-act-a-two-stage-framework","title":"Think Before You Act: A Two-Stage Framework for Mitigating Gender Bias Towards Vision-Language Tasks","date":"2024-05-27","arxiv_id":"2405.16860","repositories_listed":1,"syntology":null},{"url":"/paper/alleviating-hallucinations-in-large-vision","slug":"alleviating-hallucinations-in-large-vision","title":"Alleviating Hallucinations in Large Vision-Language Models through Hallucination-Induced Optimization","date":"2024-05-24","arxiv_id":"2405.15356","repositories_listed":1,"syntology":null},{"url":"/paper/vdgd-mitigating-lvlm-hallucinations-in","slug":"vdgd-mitigating-lvlm-hallucinations-in","title":"Visual Description Grounding Reduces Hallucinations and Boosts Reasoning in LVLMs","date":"2024-05-24","arxiv_id":"2405.15683","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vdgd-mitigating-lvlm-hallucinations-in#ran","syntology_url":"https://syntology.ai/paper/2405.15683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15683"}},"official":{"repos":["sreyan88/vdgd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/calibrated-self-rewarding-vision-language","slug":"calibrated-self-rewarding-vision-language","title":"Calibrated Self-Rewarding Vision Language Models","date":"2024-05-23","arxiv_id":"2405.14622","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/calibrated-self-rewarding-vision-language#ran","syntology_url":"https://syntology.ai/paper/2405.14622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14622"}},"official":{"repos":["yiyangzhou/csr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wise-rethinking-the-knowledge-memory-for","slug":"wise-rethinking-the-knowledge-memory-for","title":"WISE: Rethinking the Knowledge Memory for Lifelong Model Editing of Large Language Models","date":"2024-05-23","arxiv_id":"2405.14768","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":0,"n_ran_checked":4,"n_instrument":9,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 9 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/wise-rethinking-the-knowledge-memory-for#ran","syntology_url":"https://syntology.ai/paper/2405.14768","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14768"}},"official":{"repos":["zjunlp/easyedit"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/retrieval-augmented-language-model-for","slug":"retrieval-augmented-language-model-for","title":"Retrieval-Augmented Language Model for Extreme Multi-Label Knowledge Graph Link Prediction","date":"2024-05-21","arxiv_id":"2405.12656","repositories_listed":1,"syntology":null},{"url":"/paper/the-2nd-futuredial-challenge-dialog-systems","slug":"the-2nd-futuredial-challenge-dialog-systems","title":"The 2nd FutureDial Challenge: Dialog Systems with Retrieval Augmented Generation (FutureDial-RAG)","date":"2024-05-21","arxiv_id":"2405.13084","repositories_listed":1,"syntology":null},{"url":"/paper/automated-multi-level-preference-for-mllms","slug":"automated-multi-level-preference-for-mllms","title":"Automated Multi-level Preference for MLLMs","date":"2024-05-18","arxiv_id":"2405.11165","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/automated-multi-level-preference-for-mllms#ran","syntology_url":"https://syntology.ai/paper/2405.11165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11165"}},"official":{"repos":["takomc/amp"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-semantics-in-multimodal-chain-of","slug":"enhancing-semantics-in-multimodal-chain-of","title":"Enhancing Semantics in Multimodal Chain of Thought via Soft Negative Sampling","date":"2024-05-16","arxiv_id":"2405.09848","repositories_listed":1,"syntology":null},{"url":"/paper/spurious-reconstruction-from-brain-activity","slug":"spurious-reconstruction-from-brain-activity","title":"Spurious reconstruction from brain activity","date":"2024-05-16","arxiv_id":"2405.10078","repositories_listed":1,"syntology":null},{"url":"/paper/throne-an-object-based-hallucination","slug":"throne-an-object-based-hallucination","title":"THRONE: An Object-based Hallucination Benchmark for the Free-form Generations of Large Vision-Language Models","date":"2024-05-08","arxiv_id":"2405.05256","repositories_listed":1,"syntology":null},{"url":"/paper/sora-detector-a-unified-hallucination","slug":"sora-detector-a-unified-hallucination","title":"Sora Detector: A Unified Hallucination Detection for Large Text-to-Video Models","date":"2024-05-07","arxiv_id":"2405.04180","repositories_listed":1,"syntology":null},{"url":"/paper/addressing-topic-granularity-and","slug":"addressing-topic-granularity-and","title":"Addressing Topic Granularity and Hallucination in Large Language Models for Topic Modelling","date":"2024-05-01","arxiv_id":"2405.00611","repositories_listed":1,"syntology":null},{"url":"/paper/codehalu-code-hallucinations-in-llms-driven","slug":"codehalu-code-hallucinations-in-llms-driven","title":"CodeHalu: Investigating Code Hallucinations in LLMs via Execution-based Verification","date":"2024-04-30","arxiv_id":"2405.00253","repositories_listed":1,"syntology":null},{"url":"/paper/hallucination-of-multimodal-large-language","slug":"hallucination-of-multimodal-large-language","title":"Hallucination of Multimodal Large Language Models: A Survey","date":"2024-04-29","arxiv_id":"2404.18930","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-head-mechanistically-explains-long","slug":"retrieval-head-mechanistically-explains-long","title":"Retrieval Head Mechanistically Explains Long-Context Factuality","date":"2024-04-24","arxiv_id":"2404.15574","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/retrieval-head-mechanistically-explains-long#ran","syntology_url":"https://syntology.ai/paper/2404.15574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15574"}},"official":{"repos":["nightdessert/retrieval_head"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generate-on-graph-treat-llm-as-both-agent-and","slug":"generate-on-graph-treat-llm-as-both-agent-and","title":"Generate-on-Graph: Treat LLM as both Agent and KG in Incomplete Knowledge Graph Question Answering","date":"2024-04-23","arxiv_id":"2404.14741","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":11,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generate-on-graph-treat-llm-as-both-agent-and#ran","syntology_url":"https://syntology.ai/paper/2404.14741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.14741"}},"official":{"repos":["yaooxu/gog"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/detecting-and-mitigating-hallucination-in","slug":"detecting-and-mitigating-hallucination-in","title":"Detecting and Mitigating Hallucination in Large Vision Language Models via Fine-Grained AI Feedback","date":"2024-04-22","arxiv_id":"2404.14233","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/detecting-and-mitigating-hallucination-in#ran","syntology_url":"https://syntology.ai/paper/2404.14233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.14233"}},"official":{"repos":["Mr-Loevan/HSA-DPO"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/integrating-chemistry-knowledge-in-large","slug":"integrating-chemistry-knowledge-in-large","title":"Integrating Chemistry Knowledge in Large Language Models via Prompt Engineering","date":"2024-04-22","arxiv_id":"2404.14467","repositories_listed":1,"syntology":null},{"url":"/paper/llms-know-what-they-need-leveraging-a-missing","slug":"llms-know-what-they-need-leveraging-a-missing","title":"LLMs Know What They Need: Leveraging a Missing Information Guided Framework to Empower Retrieval-Augmented Generation","date":"2024-04-22","arxiv_id":"2404.14043","repositories_listed":1,"syntology":null},{"url":"/paper/valor-eval-holistic-coverage-and-faithfulness","slug":"valor-eval-holistic-coverage-and-faithfulness","title":"VALOR-EVAL: Holistic Coverage and Faithfulness Evaluation of Large Vision-Language Models","date":"2024-04-22","arxiv_id":"2404.13874","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/valor-eval-holistic-coverage-and-faithfulness#ran","syntology_url":"https://syntology.ai/paper/2404.13874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13874"}},"official":{"repos":["haoyiq114/valor"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/ai-enhanced-cognitive-behavioral-therapy-deep","slug":"ai-enhanced-cognitive-behavioral-therapy-deep","title":"AI-Enhanced Cognitive Behavioral Therapy: Deep Learning and Large Language Models for Extracting Cognitive Pathways from Social Media Texts","date":"2024-04-17","arxiv_id":"2404.11449","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-transferability-of-visual","slug":"exploring-the-transferability-of-visual","title":"Exploring the Transferability of Visual Prompting for Multimodal Large Language Models","date":"2024-04-17","arxiv_id":"2404.11207","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/exploring-the-transferability-of-visual#ran","syntology_url":"https://syntology.ai/paper/2404.11207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.11207"}},"official":{"repos":["zycheiheihei/transferable-visual-prompting"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/memllm-finetuning-llms-to-use-an-explicit","slug":"memllm-finetuning-llms-to-use-an-explicit","title":"MemLLM: Finetuning LLMs to Use An Explicit Read-Write Memory","date":"2024-04-17","arxiv_id":"2404.11672","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/memllm-finetuning-llms-to-use-an-explicit#ran","syntology_url":"https://syntology.ai/paper/2404.11672","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.11672"}},"official":{"repos":["amodaresi/memllm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-llama2-mistral-gemma-and-gpt-for","slug":"benchmarking-llama2-mistral-gemma-and-gpt-for","title":"Benchmarking Llama2, Mistral, Gemma and GPT for Factuality, Toxicity, Bias and Propensity for Hallucinations","date":"2024-04-15","arxiv_id":"2404.09785","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-llama2-mistral-gemma-and-gpt-for#ran","syntology_url":"https://syntology.ai/paper/2404.09785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09785"}},"official":{"repos":["innodatalabs/innodata-llm-safety"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/constructing-benchmarks-and-interventions-for","slug":"constructing-benchmarks-and-interventions-for","title":"Constructing Benchmarks and Interventions for Combating Hallucinations in LLMs","date":"2024-04-15","arxiv_id":"2404.09971","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constructing-benchmarks-and-interventions-for#ran","syntology_url":"https://syntology.ai/paper/2404.09971","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09971"}},"official":{"repos":["technion-cs-nlp/hallucination-mitigation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-gpt-4v-ision-for-insurance-a","slug":"harnessing-gpt-4v-ision-for-insurance-a","title":"Harnessing GPT-4V(ision) for Insurance: A Preliminary Exploration","date":"2024-04-15","arxiv_id":"2404.09690","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-hallucination-in-abstractive","slug":"mitigating-hallucination-in-abstractive","title":"Mitigating Hallucination in Abstractive Summarization with Domain-Conditional Mutual Information","date":"2024-04-15","arxiv_id":"2404.09480","repositories_listed":1,"syntology":null},{"url":"/paper/curiousllm-elevating-multi-document-qa-with","slug":"curiousllm-elevating-multi-document-qa-with","title":"CuriousLLM: Elevating Multi-Document QA with Reasoning-Infused Knowledge Graph Prompting","date":"2024-04-13","arxiv_id":"2404.09077","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-localize-objects-improves-spatial","slug":"learning-to-localize-objects-improves-spatial","title":"Learning to Localize Objects Improves Spatial Reasoning in Visual-LLMs","date":"2024-04-11","arxiv_id":"2404.07449","repositories_listed":1,"syntology":null},{"url":"/paper/smurfcat-at-semeval-2024-task-6-leveraging","slug":"smurfcat-at-semeval-2024-task-6-leveraging","title":"SmurfCat at SemEval-2024 Task 6: Leveraging Synthetic Data for Hallucination Detection","date":"2024-04-09","arxiv_id":"2404.06137","repositories_listed":1,"syntology":null},{"url":"/paper/tackling-structural-hallucination-in-image","slug":"tackling-structural-hallucination-in-image","title":"Tackling Structural Hallucination in Image Translation with Local Diffusion","date":"2024-04-09","arxiv_id":"2404.05980","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":4,"n_no_contract":5,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 4 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tackling-structural-hallucination-in-image#ran","syntology_url":"https://syntology.ai/paper/2404.05980","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05980"}},"official":{"repos":["edshkim98/localdiffusion-hallucination"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/slpl-shroom-at-semeval-2024-task-06-a","slug":"slpl-shroom-at-semeval-2024-task-06-a","title":"SLPL SHROOM at SemEval2024 Task 06: A comprehensive study on models ability to detect hallucination","date":"2024-04-07","arxiv_id":"2404.04845","repositories_listed":1,"syntology":null},{"url":"/paper/pollmgraph-unraveling-hallucinations-in-large","slug":"pollmgraph-unraveling-hallucinations-in-large","title":"PoLLMgraph: Unraveling Hallucinations in Large Language Models via State Transition Dynamics","date":"2024-04-06","arxiv_id":"2404.04722","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pollmgraph-unraveling-hallucinations-in-large#ran","syntology_url":"https://syntology.ai/paper/2404.04722","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04722"}},"official":{"repos":["hitum-dev/pollmgraph"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fakes-of-varying-shades-how-warning-affects","slug":"fakes-of-varying-shades-how-warning-affects","title":"Fakes of Varying Shades: How Warning Affects Human Perception and Engagement Regarding LLM Hallucinations","date":"2024-04-04","arxiv_id":"2404.03745","repositories_listed":1,"syntology":null},{"url":"/paper/shroom-indelab-at-semeval-2024-task-6-zero","slug":"shroom-indelab-at-semeval-2024-task-6-zero","title":"SHROOM-INDElab at SemEval-2024 Task 6: Zero- and Few-Shot LLM-Based Classification for Hallucination Detection","date":"2024-04-04","arxiv_id":"2404.03732","repositories_listed":1,"syntology":null},{"url":"/paper/knowhalu-hallucination-detection-via-multi","slug":"knowhalu-hallucination-detection-via-multi","title":"KnowHalu: Hallucination Detection via Multi-Form Knowledge Based Factual Checking","date":"2024-04-03","arxiv_id":"2404.02935","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/knowhalu-hallucination-detection-via-multi#ran","syntology_url":"https://syntology.ai/paper/2404.02935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02935"}},"official":{"repos":["javyduck/knowhalu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-model-editing-via-customized-expert","slug":"scalable-model-editing-via-customized-expert","title":"Scalable Model Editing via Customized Expert Networks","date":"2024-04-03","arxiv_id":"2404.02699","repositories_listed":1,"syntology":null},{"url":"/paper/ails-ntua-at-semeval-2024-task-6-efficient","slug":"ails-ntua-at-semeval-2024-task-6-efficient","title":"AILS-NTUA at SemEval-2024 Task 6: Efficient model tuning for hallucination detection and analysis","date":"2024-04-01","arxiv_id":"2404.01210","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-the-general-agent-capabilities-of","slug":"enhancing-the-general-agent-capabilities-of","title":"Enhancing the General Agent Capabilities of Low-Parameter LLMs through Tuning and Multi-Branch Reasoning","date":"2024-03-29","arxiv_id":"2403.19962","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/enhancing-the-general-agent-capabilities-of#ran","syntology_url":"https://syntology.ai/paper/2403.19962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19962"}},"official":{"repos":["haiv-lab/llm-tmbr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/on-large-language-models-hallucination-with","slug":"on-large-language-models-hallucination-with","title":"On Large Language Models' Hallucination with Regard to Known Facts","date":"2024-03-29","arxiv_id":"2403.20009","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-large-language-models-hallucination-with#ran","syntology_url":"https://syntology.ai/paper/2403.20009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20009"}},"official":{"repos":["dcdsf321/known_fact_hallucination"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/are-large-language-models-good-at-utility","slug":"are-large-language-models-good-at-utility","title":"Are Large Language Models Good at Utility Judgments?","date":"2024-03-28","arxiv_id":"2403.19216","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/are-large-language-models-good-at-utility#ran","syntology_url":"https://syntology.ai/paper/2403.19216","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19216"}},"official":{"repos":["ict-bigdatalab/utility_judgments"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/jdocqa-japanese-document-question-answering","slug":"jdocqa-japanese-document-question-answering","title":"JDocQA: Japanese Document Question Answering Dataset for Generative Language Models","date":"2024-03-28","arxiv_id":"2403.19454","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-correctness-without-prompting","slug":"learning-from-correctness-without-prompting","title":"Learning From Correctness Without Prompting Makes LLM Efficient Reasoner","date":"2024-03-28","arxiv_id":"2403.19094","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-from-correctness-without-prompting#ran","syntology_url":"https://syntology.ai/paper/2403.19094","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19094"}},"official":{"repos":["starrYYxuan/LeCo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/retrieval-enhanced-knowledge-editing-for","slug":"retrieval-enhanced-knowledge-editing-for","title":"Retrieval-enhanced Knowledge Editing in Language Models for Multi-Hop Question Answering","date":"2024-03-28","arxiv_id":"2403.19631","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/retrieval-enhanced-knowledge-editing-for#ran","syntology_url":"https://syntology.ai/paper/2403.19631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19631"}},"official":{"repos":["sycny/rae"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-diffusion-based-generative-equalizer-for","slug":"a-diffusion-based-generative-equalizer-for","title":"A Diffusion-Based Generative Equalizer for Music Restoration","date":"2024-03-27","arxiv_id":"2403.18636","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-diffusion-based-generative-equalizer-for#ran","syntology_url":"https://syntology.ai/paper/2403.18636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18636"}},"official":{"repos":["eloimoliner/babe2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mechanisms-of-non-factual-hallucinations-in","slug":"mechanisms-of-non-factual-hallucinations-in","title":"Mechanistic Understanding and Mitigation of Language Model Non-Factual Hallucinations","date":"2024-03-27","arxiv_id":"2403.18167","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mechanisms-of-non-factual-hallucinations-in#ran","syntology_url":"https://syntology.ai/paper/2403.18167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18167"}},"official":{"repos":["jadeleiyu/lm_hallucination_mechanisms"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-action-faithful-and-multimodal","slug":"chain-of-action-faithful-and-multimodal","title":"Chain-of-Action: Faithful and Multimodal Question Answering through Large Language Models","date":"2024-03-26","arxiv_id":"2403.17359","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chain-of-action-faithful-and-multimodal#ran","syntology_url":"https://syntology.ai/paper/2403.17359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17359"}},"official":{"repos":["MAGICS-LAB/Chain-of-Actions"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dgot-dynamic-graph-of-thoughts-for-scientific","slug":"dgot-dynamic-graph-of-thoughts-for-scientific","title":"DGoT: Dynamic Graph of Thoughts for Scientific Abstract Generation","date":"2024-03-26","arxiv_id":"2403.17491","repositories_listed":1,"syntology":null},{"url":"/paper/pensieve-retrospect-then-compare-mitigates","slug":"pensieve-retrospect-then-compare-mitigates","title":"Pensieve: Retrospect-then-Compare Mitigates Visual Hallucination","date":"2024-03-21","arxiv_id":"2403.14401","repositories_listed":1,"syntology":null},{"url":"/paper/what-if-counterfactual-inception-to-mitigate","slug":"what-if-counterfactual-inception-to-mitigate","title":"What if...?: Thinking Counterfactual Keywords Helps to Mitigate Hallucination in Large Multi-modal Models","date":"2024-03-20","arxiv_id":"2403.13513","repositories_listed":1,"syntology":null},{"url":"/paper/agent-flan-designing-data-and-methods-of","slug":"agent-flan-designing-data-and-methods-of","title":"Agent-FLAN: Designing Data and Methods of Effective Agent Tuning for Large Language Models","date":"2024-03-19","arxiv_id":"2403.12881","repositories_listed":1,"syntology":null},{"url":"/paper/logic-query-of-thoughts-guiding-large","slug":"logic-query-of-thoughts-guiding-large","title":"Logic Query of Thoughts: Guiding Large Language Models to Answer Complex Logic Queries with Knowledge Graphs","date":"2024-03-17","arxiv_id":"2404.04264","repositories_listed":1,"syntology":null},{"url":"/paper/phd-a-prompted-visual-hallucination","slug":"phd-a-prompted-visual-hallucination","title":"PhD: A ChatGPT-Prompted Visual hallucination Evaluation Dataset","date":"2024-03-17","arxiv_id":"2403.11116","repositories_listed":1,"syntology":null},{"url":"/paper/circuit-transformer-end-to-end-circuit-design","slug":"circuit-transformer-end-to-end-circuit-design","title":"Circuit Transformer: A Transformer That Preserves Logical Equivalence","date":"2024-03-14","arxiv_id":"2403.13838","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/circuit-transformer-end-to-end-circuit-design#ran","syntology_url":"https://syntology.ai/paper/2403.13838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13838"}},"official":{"repos":["snowkylin/circuit-transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/the-first-to-know-how-token-distributions","slug":"the-first-to-know-how-token-distributions","title":"The First to Know: How Token Distributions Reveal Hidden Knowledge in Large Vision-Language Models?","date":"2024-03-14","arxiv_id":"2403.09037","repositories_listed":1,"syntology":null},{"url":"/paper/xreal-realistic-anatomy-and-pathology-aware-x","slug":"xreal-realistic-anatomy-and-pathology-aware-x","title":"XReal: Realistic Anatomy and Pathology-Aware X-ray Generation via Controllable Diffusion Model","date":"2024-03-14","arxiv_id":"2403.09240","repositories_listed":1,"syntology":null},{"url":"/paper/aigcs-confuse-ai-too-investigating-and","slug":"aigcs-confuse-ai-too-investigating-and","title":"AIGCs Confuse AI Too: Investigating and Explaining Synthetic Image-induced Hallucinations in Large Vision-Language Models","date":"2024-03-13","arxiv_id":"2403.08542","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-the-performance-of-retrieval","slug":"investigating-the-performance-of-retrieval","title":"Investigating the performance of Retrieval-Augmented Generation and fine-tuning for the development of AI-driven knowledge-based systems","date":"2024-03-12","arxiv_id":"2403.09727","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-benefits-of-fine-grained-loss","slug":"on-the-benefits-of-fine-grained-loss","title":"On the Benefits of Fine-Grained Loss Truncation: A Case Study on Factuality in Summarization","date":"2024-03-09","arxiv_id":"2403.05788","repositories_listed":1,"syntology":null},{"url":"/paper/erbench-an-entity-relationship-based","slug":"erbench-an-entity-relationship-based","title":"ERBench: An Entity-Relationship based Automatically Verifiable Hallucination Benchmark for Large Language Models","date":"2024-03-08","arxiv_id":"2403.05266","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/erbench-an-entity-relationship-based#ran","syntology_url":"https://syntology.ai/paper/2403.05266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05266"}},"official":{"repos":["dilab-kaist/erbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rat-retrieval-augmented-thoughts-elicit","slug":"rat-retrieval-augmented-thoughts-elicit","title":"RAT: Retrieval Augmented Thoughts Elicit Context-Aware Reasoning in Long-Horizon Generation","date":"2024-03-08","arxiv_id":"2403.05313","repositories_listed":1,"syntology":null},{"url":"/paper/fact-checking-the-output-of-large-language","slug":"fact-checking-the-output-of-large-language","title":"Fact-Checking the Output of Large Language Models via Token-Level Uncertainty Quantification","date":"2024-03-07","arxiv_id":"2403.04696","repositories_listed":1,"syntology":null},{"url":"/paper/federated-recommendation-via-hybrid-retrieval","slug":"federated-recommendation-via-hybrid-retrieval","title":"Federated Recommendation via Hybrid Retrieval Augmented Generation","date":"2024-03-07","arxiv_id":"2403.04256","repositories_listed":1,"syntology":null},{"url":"/paper/halueval-wild-evaluating-hallucinations-of","slug":"halueval-wild-evaluating-hallucinations-of","title":"HaluEval-Wild: Evaluating Hallucinations of Language Models in the Wild","date":"2024-03-07","arxiv_id":"2403.04307","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-hallucination-in-large-language","slug":"benchmarking-hallucination-in-large-language","title":"Benchmarking Hallucination in Large Language Models based on Unanswerable Math Word Problem","date":"2024-03-06","arxiv_id":"2403.03558","repositories_listed":1,"syntology":null},{"url":"/paper/german-also-hallucinates-inconsistency","slug":"german-also-hallucinates-inconsistency","title":"German also Hallucinates! Inconsistency Detection in News Summaries with the Absinth Dataset","date":"2024-03-06","arxiv_id":"2403.03750","repositories_listed":1,"syntology":null},{"url":"/paper/in-search-of-truth-an-interrogation-approach","slug":"in-search-of-truth-an-interrogation-approach","title":"InterrogateLLM: Zero-Resource Hallucination Detection in LLM-Generated Answers","date":"2024-03-05","arxiv_id":"2403.02889","repositories_listed":1,"syntology":null},{"url":"/paper/knowagent-knowledge-augmented-planning-for","slug":"knowagent-knowledge-augmented-planning-for","title":"KnowAgent: Knowledge-Augmented Planning for LLM-Based Agents","date":"2024-03-05","arxiv_id":"2403.03101","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/knowagent-knowledge-augmented-planning-for#ran","syntology_url":"https://syntology.ai/paper/2403.03101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03101"}},"official":{"repos":["zjunlp/knowagent"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cr-lt-kgqa-a-knowledge-graph-question","slug":"cr-lt-kgqa-a-knowledge-graph-question","title":"CR-LT-KGQA: A Knowledge Graph Question Answering Dataset Requiring Commonsense Reasoning and Long-Tail Knowledge","date":"2024-03-03","arxiv_id":"2403.01395","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cr-lt-kgqa-a-knowledge-graph-question#ran","syntology_url":"https://syntology.ai/paper/2403.01395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01395"}},"official":{"repos":["d3mlab/cr-lt-kgqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/in-context-sharpness-as-alerts-an-inner","slug":"in-context-sharpness-as-alerts-an-inner","title":"In-Context Sharpness as Alerts: An Inner Representation Perspective for Hallucination Mitigation","date":"2024-03-03","arxiv_id":"2403.01548","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/in-context-sharpness-as-alerts-an-inner#ran","syntology_url":"https://syntology.ai/paper/2403.01548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.01548"}},"official":{"repos":["hkust-nlp/activation_decoding"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diahalu-a-dialogue-level-hallucination","slug":"diahalu-a-dialogue-level-hallucination","title":"DiaHalu: A Dialogue-level Hallucination Evaluation Benchmark for Large Language Models","date":"2024-03-01","arxiv_id":"2403.00896","repositories_listed":1,"syntology":null},{"url":"/paper/self-consistent-decoding-for-more-factual","slug":"self-consistent-decoding-for-more-factual","title":"Self-Consistent Decoding for More Factual Open Responses","date":"2024-03-01","arxiv_id":"2403.00696","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/self-consistent-decoding-for-more-factual#ran","syntology_url":"https://syntology.ai/paper/2403.00696","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00696"}},"official":{"repos":["cdmalon/selfconsistent"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-all-seeing-project-v2-towards-general","slug":"the-all-seeing-project-v2-towards-general","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","date":"2024-02-29","arxiv_id":"2402.19474","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":2,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-all-seeing-project-v2-towards-general#ran","syntology_url":"https://syntology.ai/paper/2402.19474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19474"}},"official":{"repos":["opengvlab/all-seeing"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/all-in-a-single-image-large-multimodal-models","slug":"all-in-a-single-image-large-multimodal-models","title":"All in an Aggregated Image for In-Image Learning","date":"2024-02-28","arxiv_id":"2402.17971","repositories_listed":1,"syntology":null},{"url":"/paper/editing-factual-knowledge-and-explanatory","slug":"editing-factual-knowledge-and-explanatory","title":"Editing Factual Knowledge and Explanatory Ability of Medical Large Language Models","date":"2024-02-28","arxiv_id":"2402.18099","repositories_listed":1,"syntology":null},{"url":"/paper/multi-fact-assessing-multilingual-llms-multi","slug":"multi-fact-assessing-multilingual-llms-multi","title":"Multi-FAct: Assessing Factuality of Multilingual LLMs using FActScore","date":"2024-02-28","arxiv_id":"2402.18045","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multi-fact-assessing-multilingual-llms-multi#ran","syntology_url":"https://syntology.ai/paper/2402.18045","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18045"}},"official":{"repos":["sheikhshafayat/multi-fact"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/re-ex-revising-after-explanation-reduces-the","slug":"re-ex-revising-after-explanation-reduces-the","title":"Re-Ex: Revising after Explanation Reduces the Factual Errors in LLM Responses","date":"2024-02-27","arxiv_id":"2402.17097","repositories_listed":1,"syntology":null},{"url":"/paper/truthx-alleviating-hallucinations-by-editing","slug":"truthx-alleviating-hallucinations-by-editing","title":"TruthX: Alleviating Hallucinations by Editing Large Language Models in Truthful Space","date":"2024-02-27","arxiv_id":"2402.17811","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/truthx-alleviating-hallucinations-by-editing#ran","syntology_url":"https://syntology.ai/paper/2402.17811","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17811"}},"official":{"repos":["ictnlp/truthx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/citation-enhanced-generation-for-llm-based","slug":"citation-enhanced-generation-for-llm-based","title":"Citation-Enhanced Generation for LLM-based Chatbots","date":"2024-02-25","arxiv_id":"2402.16063","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-machine-generated-texts-by-multi","slug":"detecting-machine-generated-texts-by-multi","title":"Detecting Machine-Generated Texts by Multi-Population Aware Optimization for Maximum Mean Discrepancy","date":"2024-02-25","arxiv_id":"2402.16041","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/detecting-machine-generated-texts-by-multi#ran","syntology_url":"https://syntology.ai/paper/2402.16041","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16041"}},"official":{"repos":["zshsh98/mmd-mp"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/hypotermqa-hypothetical-terms-dataset-for","slug":"hypotermqa-hypothetical-terms-dataset-for","title":"HypoTermQA: Hypothetical Terms Dataset for Benchmarking Hallucination Tendency of LLMs","date":"2024-02-25","arxiv_id":"2402.16211","repositories_listed":1,"syntology":null},{"url":"/paper/a-data-centric-approach-to-generate-faithful","slug":"a-data-centric-approach-to-generate-faithful","title":"A Data-Centric Approach To Generate Faithful and High Quality Patient Summaries with Large Language Models","date":"2024-02-23","arxiv_id":"2402.15422","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-data-centric-approach-to-generate-faithful#ran","syntology_url":"https://syntology.ai/paper/2402.15422","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15422"}},"official":{"repos":["stefanhgm/patient_summaries_with_llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dualfocus-integrating-macro-and-micro","slug":"dualfocus-integrating-macro-and-micro","title":"DualFocus: Integrating Macro and Micro Perspectives in Multi-modal Large Language Models","date":"2024-02-22","arxiv_id":"2402.14767","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dualfocus-integrating-macro-and-micro#ran","syntology_url":"https://syntology.ai/paper/2402.14767","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14767"}},"official":{"repos":["InternLM/InternLM-XComposer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/less-is-more-mitigating-multimodal","slug":"less-is-more-mitigating-multimodal","title":"Less is More: Mitigating Multimodal Hallucination from an EOS Decision Perspective","date":"2024-02-22","arxiv_id":"2402.14545","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":4,"n_instrument":7,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":3,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 7 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/less-is-more-mitigating-multimodal#ran","syntology_url":"https://syntology.ai/paper/2402.14545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14545"}},"official":{"repos":["yuezih/less-is-more"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ufo-a-unified-and-flexible-framework-for","slug":"ufo-a-unified-and-flexible-framework-for","title":"UFO: a Unified and Flexible Framework for Evaluating Factuality of Large Language Models","date":"2024-02-22","arxiv_id":"2402.14690","repositories_listed":1,"syntology":null}],"record_sha256":"ba9199d8a343f7fe8a274f6b292c675d34d6d018319662213eb5ec9a496c5f24","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}