{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/hallucination/papers/14","list_of":"/task/hallucination","task":"Hallucination","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":14,"pages_in_order":19,"rows_per_page":100,"rows":[1301,1400],"of":1816,"counts":{"archive_papers_tagged":1816,"with_a_code_link":752,"where_syntology_ran_a_sample":276,"not_listed_spam_title":0,"listed":1816,"listed_where_code_ran":276,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":240,"every_run_a_failure_of_syntologys_instrument":36,"listed_with_a_run_with_no_instrument_failure":240,"listed_every_run_a_failure_of_syntologys_instrument":36,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/hallucination","prev":"/task/hallucination/papers/13","next":"/task/hallucination/papers/15","papers":[{"url":null,"slug":"wildhallucinations-evaluating-long-form","title":"WildHallucinations: Evaluating Long-form Factuality in LLMs with Real-World Entity Queries","date":"2024-07-24","arxiv_id":"2407.17468","repositories_listed":0,"syntology":null},{"url":null,"slug":"generation-constraint-scaling-can-mitigate","title":"Generation Constraint Scaling Can Mitigate Hallucination","date":"2024-07-23","arxiv_id":"2407.16908","repositories_listed":0,"syntology":null},{"url":null,"slug":"lawluo-a-chinese-law-firm-co-run-by-llm","title":"LawLuo: A Multi-Agent Collaborative Framework for Multi-Round Chinese Legal Consultation","date":"2024-07-23","arxiv_id":"2407.16252","repositories_listed":0,"syntology":null},{"url":null,"slug":"shared-imagination-llms-hallucinate-alike","title":"Shared Imagination: LLMs Hallucinate Alike","date":"2024-07-23","arxiv_id":"2407.16604","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-a-reliable-general-purpose","title":"Developing a Reliable, Fast, General-Purpose Hallucination Detection and Mitigation Service","date":"2024-07-22","arxiv_id":"2407.15441","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-fine-grained-news-headline","title":"Multilingual Fine-Grained News Headline Hallucination Detection","date":"2024-07-22","arxiv_id":"2407.15975","repositories_listed":0,"syntology":null},{"url":null,"slug":"text2place-affordance-aware-text-guided-human","title":"Text2Place: Affordance-aware Text Guided Human Placement","date":"2024-07-22","arxiv_id":"2407.15446","repositories_listed":0,"syntology":null},{"url":null,"slug":"2408-01433","title":"Evaluating and Enhancing Trustworthiness of LLMs in Perception Tasks","date":"2024-07-18","arxiv_id":"2408.01433","repositories_listed":0,"syntology":null},{"url":null,"slug":"beaf-observing-before-after-changes-to","title":"BEAF: Observing BEfore-AFter Changes to Evaluate Hallucination in Vision-language Models","date":"2024-07-18","arxiv_id":"2407.13442","repositories_listed":0,"syntology":null},{"url":null,"slug":"black-box-opinion-manipulation-attacks-to","title":"Black-Box Opinion Manipulation Attacks to Retrieval-Augmented Generation of Large Language Models","date":"2024-07-18","arxiv_id":"2407.13757","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-augmented-generation-for-natural","title":"Retrieval-Augmented Generation for Natural Language Processing: A Survey","date":"2024-07-18","arxiv_id":"2407.13193","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-image-hallucination-in-text-to","title":"Addressing Image Hallucination in Text-to-Image Generation through Factual Image Retrieval","date":"2024-07-15","arxiv_id":"2407.10683","repositories_listed":0,"syntology":null},{"url":null,"slug":"grapheval-a-knowledge-graph-based-llm","title":"GraphEval: A Knowledge-Graph Based LLM Hallucination Evaluation Framework","date":"2024-07-15","arxiv_id":"2407.10793","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-within-why-llms-hallucinate-a-causal","title":"Look Within, Why LLMs Hallucinate: A Causal Perspective","date":"2024-07-14","arxiv_id":"2407.10153","repositories_listed":0,"syntology":null},{"url":null,"slug":"cohesive-conversations-enhancing-authenticity","title":"Cohesive Conversations: Enhancing Authenticity in Multi-Agent Simulated Dialogues","date":"2024-07-13","arxiv_id":"2407.09897","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-mitigating-code-llm-hallucinations-with","title":"On Mitigating Code LLM Hallucinations with API Documentation","date":"2024-07-13","arxiv_id":"2407.09726","repositories_listed":0,"syntology":null},{"url":null,"slug":"dahrs-divergence-aware-hallucination","title":"DAHRS: Divergence-Aware Hallucination-Remediated SRL Projection","date":"2024-07-12","arxiv_id":"2407.09283","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-two-sides-of-the-coin-hallucination","title":"The Two Sides of the Coin: Hallucination Generation and Detection with LLMs as Evaluators for LLMs","date":"2024-07-12","arxiv_id":"2407.09152","repositories_listed":0,"syntology":null},{"url":null,"slug":"lynx-an-open-source-hallucination-evaluation","title":"Lynx: An Open Source Hallucination Evaluation Model","date":"2024-07-11","arxiv_id":"2407.08488","repositories_listed":0,"syntology":null},{"url":null,"slug":"fuse-reason-and-verify-geometry-problem","title":"Fuse, Reason and Verify: Geometry Problem Solving with Parsed Clauses from Diagram","date":"2024-07-10","arxiv_id":"2407.07327","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-overshadowing-causes-amalgamated","title":"Knowledge Overshadowing Causes Amalgamated Hallucination in Large Language Models","date":"2024-07-10","arxiv_id":"2407.08039","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-with-instance-dependent-noisy-labels","title":"Learning with Instance-Dependent Noisy Labels by Anchor Hallucination and Hard Sample Label Correction","date":"2024-07-10","arxiv_id":"2407.07331","repositories_listed":0,"syntology":null},{"url":null,"slug":"gtp-4o-modality-prompted-heterogeneous-graph","title":"GTP-4o: Modality-prompted Heterogeneous Graph Learning for Omni-modal Biomedical Representation","date":"2024-07-08","arxiv_id":"2407.05540","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-language-models-under-cultural-and","title":"Vision-Language Models under Cultural and Inclusive Considerations","date":"2024-07-08","arxiv_id":"2407.06177","repositories_listed":0,"syntology":null},{"url":null,"slug":"videocot-a-video-chain-of-thought-dataset","title":"VideoCoT: A Video Chain-of-Thought Dataset with Active Annotation Tool","date":"2024-07-07","arxiv_id":"2407.05355","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-hallucination","title":"Code Hallucination","date":"2024-07-05","arxiv_id":"2407.04831","repositories_listed":0,"syntology":null},{"url":null,"slug":"classification-based-automatic-hdl-code","title":"Classification-Based Automatic HDL Code Generation Using LLMs","date":"2024-07-04","arxiv_id":"2407.18326","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-detection-robustly-discerning","title":"Hallucination Detection: Robustly Discerning Reliable Answers in Large Language Models","date":"2024-07-04","arxiv_id":"2407.04121","repositories_listed":0,"syntology":null},{"url":null,"slug":"query-guided-self-supervised-summarization-of","title":"Query-Guided Self-Supervised Summarization of Nursing Notes","date":"2024-07-04","arxiv_id":"2407.04125","repositories_listed":0,"syntology":null},{"url":null,"slug":"stoc-tot-stochastic-tree-of-thought-with","title":"STOC-TOT: Stochastic Tree-of-Thought with Constrained Decoding for Complex Reasoning in Multi-Hop Question Answering","date":"2024-07-04","arxiv_id":"2407.03687","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-persuasive-chatbots-with-llm","title":"Zero-shot Persuasive Chatbots with LLM-Generated Strategies and Information Retrieval","date":"2024-07-04","arxiv_id":"2407.03585","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-study-of-dsl-code-generation","title":"A Comparative Study of DSL Code Generation: Fine-Tuning vs. Optimized Retrieval Augmentation","date":"2024-07-03","arxiv_id":"2407.02742","repositories_listed":0,"syntology":null},{"url":null,"slug":"fsm-a-finite-state-machine-based-zero-shot","title":"FSM: A Finite State Machine Based Zero-Shot Prompting Paradigm for Multi-Hop Question Answering","date":"2024-07-03","arxiv_id":"2407.02964","repositories_listed":0,"syntology":null},{"url":null,"slug":"pelican-correcting-hallucination-in-vision","title":"Pelican: Correcting Hallucination in Vision-LLMs via Claim Decomposition and Program of Thought Verification","date":"2024-07-02","arxiv_id":"2407.02352","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-alignment-in-multimodal-llms-a","title":"Understanding Alignment in Multimodal LLMs: A Comprehensive Study","date":"2024-07-02","arxiv_id":"2407.02477","repositories_listed":0,"syntology":null},{"url":null,"slug":"free-text-rationale-generation-under","title":"Free-text Rationale Generation under Readability Level Control","date":"2024-07-01","arxiv_id":"2407.01384","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-uncertainty-quantification-through","title":"LLM Uncertainty Quantification through Directional Entailment Graph and Claim Level Response Augmentation","date":"2024-07-01","arxiv_id":"2407.00994","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-need-for-guardrails-with-large-language","title":"The Need for Guardrails with Large Language Models in Medical Safety-Critical Settings: An Artificial Intelligence Application in the Pharmacovigilance Ecosystem","date":"2024-07-01","arxiv_id":"2407.18322","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-glitches-a-deep-dive-into-image","title":"Unveiling Glitches: A Deep Dive into Image Encoding Bugs within CLIP","date":"2024-06-30","arxiv_id":"2407.00592","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-on-effect-of-reference-knowledge","title":"A Study on Effect of Reference Knowledge Choice in Generating Technical Content Relevant to SAPPhIRE Model Using Large Language Model","date":"2024-06-29","arxiv_id":"2407.00396","repositories_listed":0,"syntology":null},{"url":null,"slug":"pfme-a-modular-approach-for-fine-grained","title":"PFME: A Modular Approach for Fine-grained Hallucination Detection and Editing of Large Language Models","date":"2024-06-29","arxiv_id":"2407.00488","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-rlaif-for-code-generation-with-api","title":"Applying RLAIF for Code Generation with API-usage in Lightweight LLMs","date":"2024-06-28","arxiv_id":"2406.20060","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-consistency-image-generation-pcig-a","title":"Prompt-Consistency Image Generation (PCIG): A Unified Framework Integrating LLMs, Knowledge Graphs, and Controllable Diffusion Models","date":"2024-06-24","arxiv_id":"2406.16333","repositories_listed":0,"syntology":null},{"url":null,"slug":"videohallucer-evaluating-intrinsic-and","title":"VideoHallucer: Evaluating Intrinsic and Extrinsic Hallucinations in Large Video-Language Models","date":"2024-06-24","arxiv_id":"2406.16338","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-object-grounding-really-reduce","title":"Does Object Grounding Really Reduce Hallucination of Large Vision-Language Models?","date":"2024-06-20","arxiv_id":"2406.14492","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-descriptive-richness-to-bias-unveiling","title":"From Descriptive Richness to Bias: Unveiling the Dark Side of Generative Image Caption Enrichment","date":"2024-06-20","arxiv_id":"2406.13912","repositories_listed":0,"syntology":null},{"url":null,"slug":"hight-hierarchical-graph-tokenization-for","title":"HIGHT: Hierarchical Graph Tokenization for Molecule-Language Alignment","date":"2024-06-20","arxiv_id":"2406.14021","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-are-skeptics-false","title":"Large Language Models are Skeptics: False Negative Problem of Input-conflicting Hallucination","date":"2024-06-20","arxiv_id":"2406.13929","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-under-alignment-atomic-preference","title":"Beyond Under-Alignment: Atomic Preference Enhanced Factuality Tuning for Large Language Models","date":"2024-06-18","arxiv_id":"2406.12416","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-more-details-always-introduce-more","title":"Do More Details Always Introduce More Hallucinations in LVLM-based Image Captioning?","date":"2024-06-18","arxiv_id":"2406.12663","repositories_listed":0,"syntology":null},{"url":null,"slug":"richrag-crafting-rich-responses-for-multi","title":"RichRAG: Crafting Rich Responses for Multi-faceted Queries in Retrieval-Augmented Generation","date":"2024-06-18","arxiv_id":"2406.12566","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-matters-in-learning-facts-in-language","title":"What Matters in Memorizing and Recalling Facts? Multifaceted Benchmarks for Knowledge Probing in Language Models","date":"2024-06-18","arxiv_id":"2406.12277","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-mitigation-prompts-long-term","title":"Hallucination Mitigation Prompts Long-term Video Understanding","date":"2024-06-17","arxiv_id":"2406.11333","repositories_listed":0,"syntology":null},{"url":null,"slug":"internalinspector-i-2-robust-confidence","title":"InternalInspector $I^2$: Robust Confidence Estimation in LLMs through Internal States","date":"2024-06-17","arxiv_id":"2406.12053","repositories_listed":0,"syntology":null},{"url":null,"slug":"medthink-inducing-medical-large-scale-visual","title":"CoMT: Chain-of-Medical-Thought Reduces Hallucination in Medical Report Generation","date":"2024-06-17","arxiv_id":"2406.11451","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-large-language-model-hallucination","title":"Mitigating Large Language Model Hallucination with Faithful Finetuning","date":"2024-06-17","arxiv_id":"2406.11267","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-large-language-models-to-express","title":"Teaching Large Language Models to Express Knowledge Boundary from Their Own Signals","date":"2024-06-16","arxiv_id":"2406.10881","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-and-evaluating-medical","title":"Detecting and Evaluating Medical Hallucinations in Large Vision Language Models","date":"2024-06-14","arxiv_id":"2406.10185","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-words-on-large-language-models","title":"Beyond Words: On Large Language Models Actionability in Mission-Critical Risk Analysis","date":"2024-06-11","arxiv_id":"2406.10273","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-the-hallucination-rate-of","title":"Estimating the Hallucination Rate of Generative AI","date":"2024-06-11","arxiv_id":"2406.07457","repositories_listed":0,"syntology":null},{"url":null,"slug":"halludial-a-large-scale-benchmark-for","title":"HalluDial: A Large-Scale Benchmark for Automatic Dialogue-Level Hallucination Evaluation","date":"2024-06-11","arxiv_id":"2406.07070","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-query-expansion-for-retrieval","title":"Progressive Query Expansion for Retrieval Over Cost-constrained Data Sources","date":"2024-06-11","arxiv_id":"2406.07136","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-and-addressing-hallucinations","title":"Investigating and Addressing Hallucinations of LLMs in Tasks Involving Negation","date":"2024-06-08","arxiv_id":"2406.05494","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-assessment-of-mathematical","title":"Robustness Assessment of Mathematical Reasoning in the Presence of Missing and Contradictory Conditions","date":"2024-06-07","arxiv_id":"2406.05055","repositories_listed":0,"syntology":null},{"url":null,"slug":"actionreasoningbench-reasoning-about-actions","title":"ActionReasoningBench: Reasoning about Actions with and without Ramification Constraints","date":"2024-06-06","arxiv_id":"2406.04046","repositories_listed":0,"syntology":null},{"url":null,"slug":"chaos-with-keywords-exposing-large-language","title":"Chaos with Keywords: Exposing Large Language Models Sycophantic Hallucination to Misleading Keywords and Evaluating Defense Strategies","date":"2024-06-06","arxiv_id":"2406.03827","repositories_listed":0,"syntology":null},{"url":null,"slug":"confabulation-the-surprising-value-of-large","title":"Confabulation: The Surprising Value of Large Language Model Hallucinations","date":"2024-06-06","arxiv_id":"2406.04175","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-llm-behavior-in-dialogue","title":"Analyzing LLM Behavior in Dialogue Summarization: Unveiling Circumstantial Hallucination Trends","date":"2024-06-05","arxiv_id":"2406.03487","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-detecting-llms-hallucination-via","title":"Towards Detecting LLMs Hallucination via Markov Chain-based Multi-agent Debate Framework","date":"2024-06-05","arxiv_id":"2406.03075","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-contrasting-self-generated-description","title":"CODE: Contrasting Self-generated Description to Combat Hallucination in Large Multi-modal Models","date":"2024-06-04","arxiv_id":"2406.01920","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-trust-in-llms-algorithms-for","title":"Enhancing Trust in LLMs: Algorithms for Comparing and Interpreting LLMs","date":"2024-06-04","arxiv_id":"2406.01943","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-explore-with-belief-state-entropy","title":"How to Explore with Belief: State Entropy Maximization in POMDPs","date":"2024-06-04","arxiv_id":"2406.02295","repositories_listed":0,"syntology":null},{"url":null,"slug":"ask-eda-a-design-assistant-empowered-by-llm","title":"Ask-EDA: A Design Assistant Empowered by LLM, Hybrid RAG and Abbreviation De-hallucination","date":"2024-06-03","arxiv_id":"2406.06575","repositories_listed":0,"syntology":null},{"url":null,"slug":"decompose-enrich-and-extract-schema-aware","title":"Decompose, Enrich, and Extract! Schema-aware Event Extraction using LLMs","date":"2024-06-03","arxiv_id":"2406.01045","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-assisted-optimal-bidding","title":"Large Language Model Assisted Optimal Bidding of BESS in FCAS Market: An AI-agent based Approach","date":"2024-06-03","arxiv_id":"2406.00974","repositories_listed":0,"syntology":null},{"url":null,"slug":"luna-an-evaluation-foundation-model-to-catch","title":"Luna: An Evaluation Foundation Model to Catch Language Model Hallucinations with High Accuracy and Low Cost","date":"2024-06-03","arxiv_id":"2406.00975","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-modeling-for-short-texts-with-large","title":"Comprehensive Evaluation of Large Language Models for Topic Modeling","date":"2024-06-02","arxiv_id":"2406.00697","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-free-assessing-the-reliability","title":"Hallucination-Free? Assessing the Reliability of Leading AI Legal Research Tools","date":"2024-05-30","arxiv_id":"2405.20362","repositories_listed":0,"syntology":null},{"url":null,"slug":"similarity-is-not-all-you-need-endowing","title":"Similarity is Not All You Need: Endowing Retrieval Augmented Generation with Multi Layered Thoughts","date":"2024-05-30","arxiv_id":"2405.19893","repositories_listed":0,"syntology":null},{"url":null,"slug":"massive-multilingual-abstract-meaning","title":"MASSIVE Multilingual Abstract Meaning Representation: A Dataset and Baselines for Hallucination Detection","date":"2024-05-29","arxiv_id":"2405.19285","repositories_listed":0,"syntology":null},{"url":null,"slug":"metatoken-detecting-hallucination-in-image","title":"MetaToken: Detecting Hallucination in Image Descriptions by Meta Classification","date":"2024-05-29","arxiv_id":"2405.19186","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-layer-retrieval-augmented-generation","title":"Two-Layer Retrieval-Augmented Generation Framework for Low-Resource Medical Question Answering Using Reddit Data: Proof-of-Concept Study","date":"2024-05-29","arxiv_id":"2405.19519","repositories_listed":0,"syntology":null},{"url":null,"slug":"conv-coa-improving-open-domain-question","title":"Conv-CoA: Improving Open-domain Question Answering in Large Language Models via Conversational Chain-of-Action","date":"2024-05-28","arxiv_id":"2405.17822","repositories_listed":0,"syntology":null},{"url":"/paper/mitigating-object-hallucination-via-data","slug":"mitigating-object-hallucination-via-data","title":"Data-augmented phrase-level alignment for mitigating object hallucination","date":"2024-05-28","arxiv_id":"2405.18654","repositories_listed":0,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mitigating-object-hallucination-via-data#ran","syntology_url":"https://syntology.ai/paper/2405.18654","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18654"}},"official":null}},{"url":null,"slug":"ritual-random-image-transformations-as-a","title":"RITUAL: Random Image Transformations as a Universal Anti-hallucination Lever in Large Vision Language Models","date":"2024-05-28","arxiv_id":"2405.17821","repositories_listed":0,"syntology":null},{"url":null,"slug":"laboratory-scale-ai-open-weight-models-are","title":"Laboratory-Scale AI: Open-Weight Models are Competitive with ChatGPT Even in Low-Resource Settings","date":"2024-05-27","arxiv_id":"2405.16820","repositories_listed":0,"syntology":null},{"url":null,"slug":"geneagent-self-verification-language-agent","title":"GeneAgent: Self-verification Language Agent for Gene Set Knowledge Discovery using Domain Databases","date":"2024-05-25","arxiv_id":"2405.16205","repositories_listed":0,"syntology":null},{"url":null,"slug":"charp-conversation-history-awareness-probing","title":"CHARP: Conversation History AwaReness Probing for Knowledge-grounded Dialogue Systems","date":"2024-05-24","arxiv_id":"2405.15110","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-model-pruning","title":"Large Language Model Pruning","date":"2024-05-24","arxiv_id":"2406.00030","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-laws-for-discriminative","title":"Scaling Laws for Discriminative Classification in Large Language Models","date":"2024-05-24","arxiv_id":"2405.15765","repositories_listed":0,"syntology":null},{"url":null,"slug":"crosscheckgpt-universal-hallucination-ranking","title":"CrossCheckGPT: Universal Hallucination Ranking for Multimodal Foundation Models","date":"2024-05-22","arxiv_id":"2405.13684","repositories_listed":0,"syntology":null},{"url":null,"slug":"feedback-aligned-mixed-llms-for-machine","title":"Less for More: Enhanced Feedback-aligned Mixed LLMs for Molecule Caption Generation and Fine-Grained NLI Evaluation","date":"2024-05-22","arxiv_id":"2405.13984","repositories_listed":0,"syntology":null},{"url":null,"slug":"gamevlm-a-decision-making-framework-for","title":"GameVLM: A Decision-making Framework for Robotic Task Planning Based on Visual Language Models and Zero-sum Games","date":"2024-05-22","arxiv_id":"2405.13751","repositories_listed":0,"syntology":null},{"url":null,"slug":"gradient-projection-for-parameter-efficient","title":"Gradient Projection For Continual Parameter-Efficient Tuning","date":"2024-05-22","arxiv_id":"2405.13383","repositories_listed":0,"syntology":null},{"url":null,"slug":"presentations-are-not-always-linear-gnn-meets","title":"Presentations are not always linear! GNN meets LLM for Document-to-Presentation Transformation with Attribution","date":"2024-05-21","arxiv_id":"2405.13095","repositories_listed":0,"syntology":null},{"url":null,"slug":"ct-eval-benchmarking-chinese-text-to-table","title":"CT-Eval: Benchmarking Chinese Text-to-Table Performance in Large Language Models","date":"2024-05-20","arxiv_id":"2405.12174","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-text-to-speech-synthesis-from-a","title":"Evaluating Text-to-Speech Synthesis from a Large Discrete Token-based Speech Language Model","date":"2024-05-16","arxiv_id":"2405.09768","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-hallucination-in-text-image-video","title":"A Comprehensive Survey of Hallucination in Large Language, Image, Video and Audio Foundation Models","date":"2024-05-15","arxiv_id":"2405.09589","repositories_listed":0,"syntology":null},{"url":null,"slug":"word-alignment-as-preference-for-machine","title":"Word Alignment as Preference for Machine Translation","date":"2024-05-15","arxiv_id":"2405.09223","repositories_listed":0,"syntology":null},{"url":null,"slug":"almol-aligned-language-molecule-translation","title":"ALMol: Aligned Language-Molecule Translation LLMs through Offline Preference Contrastive Optimisation","date":"2024-05-14","arxiv_id":"2405.08619","repositories_listed":0,"syntology":null}],"record_sha256":"d891c9bb1c810753a2e77fb3855234125cf6e4fb5dc182667ee1d9b7c24efdce","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}