{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/45","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":45,"pages_in_order":109,"rows_per_page":100,"rows":[4401,4500],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/44","next":"/task/question-answering/papers/46","papers":[{"url":null,"slug":"ccnu-at-semeval-2025-task-3-leveraging","title":"CCNU at SemEval-2025 Task 3: Leveraging Internal and External Knowledge of Large Language Models for Multilingual Hallucination Annotation","date":"2025-05-17","arxiv_id":"2505.11965","repositories_listed":0,"syntology":null},{"url":null,"slug":"recursive-question-understanding-for-complex","title":"Recursive Question Understanding for Complex Question Answering over Heterogeneous Personal Data","date":"2025-05-17","arxiv_id":"2505.11900","repositories_listed":0,"syntology":null},{"url":null,"slug":"tinyrs-r1-compact-multimodal-language-model","title":"TinyRS-R1: Compact Multimodal Language Model for Remote Sensing","date":"2025-05-17","arxiv_id":"2505.12099","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveiling-knowledge-utilization-mechanisms-in","title":"Unveiling Knowledge Utilization Mechanisms in LLM-based Retrieval-Augmented Generation","date":"2025-05-17","arxiv_id":"2505.11995","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11079","title":"$\\mathcal{A}LLM4ADD$: Unlocking the Capabilities of Audio Large Language Models for Audio Deepfake Detection","date":"2025-05-16","arxiv_id":"2505.11079","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11271","title":"Semantic Caching of Contextual Summaries for Efficient Question-Answering with Language Models","date":"2025-05-16","arxiv_id":"2505.11271","repositories_listed":0,"syntology":null},{"url":null,"slug":"thelma-task-based-holistic-evaluation-of","title":"THELMA: Task Based Holistic Evaluation of Large Language Model Applications-RAG Question Answering","date":"2025-05-16","arxiv_id":"2505.11626","repositories_listed":0,"syntology":null},{"url":null,"slug":"cafe-retrieval-head-based-coarse-to-fine","title":"CAFE: Retrieval Head-based Coarse-to-Fine Information Seeking to Enhance Multi-Document QA Capability","date":"2025-05-15","arxiv_id":"2505.10063","repositories_listed":0,"syntology":null},{"url":null,"slug":"dif-a-framework-for-benchmarking-and","title":"DIF: A Framework for Benchmarking and Verifying Implicit Bias in LLMs","date":"2025-05-15","arxiv_id":"2505.10013","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-vision-tokenizer-tuning","title":"End-to-End Vision Tokenizer Tuning","date":"2025-05-15","arxiv_id":"2505.10562","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-multi-image-question-answering-via","title":"Enhancing Multi-Image Question Answering via Submodular Subset Selection","date":"2025-05-15","arxiv_id":"2505.10533","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-graph-retrieval-augmented","title":"Leveraging Graph Retrieval-Augmented Generation to Support Learners' Understanding of Knowledge Concepts in MOOCs","date":"2025-05-15","arxiv_id":"2505.10074","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-does-neuro-mean-to-cardio-investigating","title":"What Does Neuro Mean to Cardio? Investigating the Role of Clinical Specialty Data in Medical LLMs","date":"2025-05-15","arxiv_id":"2505.10113","repositories_listed":0,"syntology":null},{"url":null,"slug":"omni-r1-do-you-really-need-audio-to-fine-tune","title":"Omni-R1: Do You Really Need Audio to Fine-Tune Your Audio LLM?","date":"2025-05-14","arxiv_id":"2505.09439","repositories_listed":0,"syntology":null},{"url":null,"slug":"safepath-conformal-prediction-for-safe-llm","title":"SafePath: Conformal Prediction for Safe LLM-Based Autonomous Navigation","date":"2025-05-14","arxiv_id":"2505.09427","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-large-language-models-on-task","title":"The Impact of Large Language Models on Task Automation in Manufacturing Services","date":"2025-05-14","arxiv_id":"2505.10581","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-visual-question-answering","title":"Variational Visual Question Answering","date":"2025-05-14","arxiv_id":"2505.09591","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusing-bidirectional-chains-of-thought-and","title":"Fusing Bidirectional Chains of Thought and Reward Mechanisms A Method for Enhancing Question-Answering Capabilities of Large Language Models for Chinese Intangible Cultural Heritage","date":"2025-05-13","arxiv_id":"2505.08167","repositories_listed":0,"syntology":null},{"url":null,"slug":"wixqa-a-multi-dataset-benchmark-for","title":"WixQA: A Multi-Dataset Benchmark for Enterprise Retrieval-Augmented Generation","date":"2025-05-13","arxiv_id":"2505.08643","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-domain-audio-question-answering-toward","title":"Multi-Domain Audio Question Answering Toward Acoustic Content Reasoning in The DCASE 2025 Challenge","date":"2025-05-12","arxiv_id":"2505.07365","repositories_listed":0,"syntology":null},{"url":null,"slug":"private-lora-fine-tuning-of-open-source-llms","title":"Private LoRA Fine-tuning of Open-Source LLMs with Homomorphic Encryption","date":"2025-05-12","arxiv_id":"2505.07329","repositories_listed":0,"syntology":null},{"url":null,"slug":"relative-overfitting-and-accept-reject","title":"Relative Overfitting and Accept-Reject Framework","date":"2025-05-12","arxiv_id":"2505.07783","repositories_listed":0,"syntology":null},{"url":null,"slug":"visually-interpretable-subtask-reasoning-for","title":"Visually Interpretable Subtask Reasoning for Visual Question Answering","date":"2025-05-12","arxiv_id":"2505.08084","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-human-verified-clinical-reasoning","title":"Building a Human-Verified Clinical Reasoning Dataset via a Human LLM Hybrid Pipeline for Trustworthy Medical AI","date":"2025-05-11","arxiv_id":"2505.06912","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-explainable-medical-ai-assistant","title":"Multi-Modal Explainable Medical AI Assistant for Trustworthy Human-AI Collaboration","date":"2025-05-11","arxiv_id":"2505.06898","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-the-nlpcc-2025-shared-task-4","title":"Overview of the NLPCC 2025 Shared Task 4: Multi-modal, Multilingual, and Multi-hop Medical Instructional Video Question Answering Challenge","date":"2025-05-11","arxiv_id":"2505.06814","repositories_listed":0,"syntology":null},{"url":null,"slug":"plhf-prompt-optimization-with-few-shot-human","title":"PLHF: Prompt Optimization with Few-Shot Human Feedback","date":"2025-05-11","arxiv_id":"2505.07886","repositories_listed":0,"syntology":null},{"url":"/paper/omgm-orchestrate-multiple-granularities-and","slug":"omgm-orchestrate-multiple-granularities-and","title":"OMGM: Orchestrate Multiple Granularities and Modalities for Efficient Multimodal Retrieval","date":"2025-05-10","arxiv_id":"2505.07879","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/omgm-orchestrate-multiple-granularities-and#ran","syntology_url":"https://syntology.ai/paper/2505.07879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07879"}},"official":null}},{"url":null,"slug":"a-grounded-memory-system-for-smart-personal","title":"A Grounded Memory System For Smart Personal Assistants","date":"2025-05-09","arxiv_id":"2505.06328","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-robustness-to-spurious-correlations","title":"Assessing Robustness to Spurious Correlations in Post-Training Language Models","date":"2025-05-09","arxiv_id":"2505.05704","repositories_listed":0,"syntology":null},{"url":null,"slug":"cellverse-do-large-language-models-really","title":"CellVerse: Do Large Language Models Really Understand Cell Biology?","date":"2025-05-09","arxiv_id":"2505.07865","repositories_listed":0,"syntology":null},{"url":null,"slug":"document-attribution-examining-citation","title":"Document Attribution: Examining Citation Relationships using Large Language Models","date":"2025-05-09","arxiv_id":"2505.06324","repositories_listed":0,"syntology":null},{"url":null,"slug":"healthy-llms-benchmarking-llm-knowledge-of-uk","title":"Healthy LLMs? Benchmarking LLM Knowledge of UK Government Public Health Information","date":"2025-05-09","arxiv_id":"2505.06046","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-reflection-backdoor-attack-on-vision","title":"Natural Reflection Backdoor Attack on Vision Language Model for Autonomous Driving","date":"2025-05-09","arxiv_id":"2505.06413","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-developmentally-plausible-rewards","title":"Towards Developmentally Plausible Rewards: Communicative Success as a Learning Signal for Interactive Language Models","date":"2025-05-09","arxiv_id":"2505.05970","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-open-source-dual-loss-embedding-model-for","title":"An Open-Source Dual-Loss Embedding Model for Semantic Retrieval in Higher Education","date":"2025-05-08","arxiv_id":"2505.04916","repositories_listed":0,"syntology":null},{"url":null,"slug":"lost-in-ocr-translation-vision-based","title":"Lost in OCR Translation? Vision-Based Approaches to Robust Document Retrieval","date":"2025-05-08","arxiv_id":"2505.05666","repositories_listed":0,"syntology":null},{"url":null,"slug":"site-towards-spatial-intelligence-thorough","title":"SITE: towards Spatial Intelligence Thorough Evaluation","date":"2025-05-08","arxiv_id":"2505.05456","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-large-language-models-and","title":"Fine-Tuning Large Language Models and Evaluating Retrieval Methods for Improved Question Answering on Building Codes","date":"2025-05-07","arxiv_id":"2505.04666","repositories_listed":0,"syntology":null},{"url":null,"slug":"hiperrag-high-performance-retrieval-augmented","title":"HiPerRAG: High-Performance Retrieval Augmented Generation for Scientific Insights","date":"2025-05-07","arxiv_id":"2505.04846","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-heart-ecg-question-answering-via-knowledge","title":"Q-Heart: ECG Question Answering via Knowledge-Informed Multimodal LLMs","date":"2025-05-07","arxiv_id":"2505.06296","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reasoning-focused-legal-retrieval-benchmark","title":"A Reasoning-Focused Legal Retrieval Benchmark","date":"2025-05-06","arxiv_id":"2505.03970","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlm-q-learning-aligning-vision-language","title":"VLM Q-Learning: Aligning Vision-Language Models for Interactive Decision-Making","date":"2025-05-06","arxiv_id":"2505.03181","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-causal-models-and-llms-integration","title":"Structure Causal Models and LLMs Integration in Medical Visual Question Answering","date":"2025-05-05","arxiv_id":"2505.02703","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-semantic-communication-in-large","title":"Task-Oriented Semantic Communication in Large Multimodal Models-based Vehicle Networks","date":"2025-05-05","arxiv_id":"2505.02413","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-image-text-matching-and","title":"Compositional Image-Text Matching and Retrieval by Grounding Entities","date":"2025-05-04","arxiv_id":"2505.02278","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-token-boundaries-integrating-human","title":"Adaptive Token Boundaries: Integrating Human Chunking Mechanisms into Multimodal LLMs","date":"2025-05-03","arxiv_id":"2505.04637","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-augmented-language-models","title":"Knowledge-Augmented Language Models Interpreting Structured Chest X-Ray Findings","date":"2025-05-03","arxiv_id":"2505.01711","repositories_listed":0,"syntology":null},{"url":null,"slug":"oodte-a-differential-testing-engine-for-the","title":"OODTE: A Differential Testing Engine for the ONNX Optimizer","date":"2025-05-03","arxiv_id":"2505.01892","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-attention-toward-machines-with","title":"Beyond Attention: Toward Machines with Intrinsic Higher Mental States","date":"2025-05-02","arxiv_id":"2505.06257","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-task-assistance-with-multimodal","title":"Grounding Task Assistance with Multimodal Cues from a Single Demonstration","date":"2025-05-02","arxiv_id":"2505.01578","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-adversarial-attacks-on-black-box","title":"Transferable Adversarial Attacks on Black-Box Vision-Language Models","date":"2025-05-02","arxiv_id":"2505.01050","repositories_listed":0,"syntology":null},{"url":null,"slug":"traveler-a-benchmark-for-evaluating-temporal","title":"TRAVELER: A Benchmark for Evaluating Temporal Reasoning across Vague, Implicit and Explicit References","date":"2025-05-02","arxiv_id":"2505.01325","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallumix-a-task-agnostic-multi-domain","title":"HalluMix: A Task-Agnostic, Multi-Domain Benchmark for Real-World Hallucination Detection","date":"2025-05-01","arxiv_id":"2505.00506","repositories_listed":0,"syntology":null},{"url":null,"slug":"calibrating-uncertainty-quantification-of","title":"Calibrating Uncertainty Quantification of Multi-Modal LLMs using Grounding","date":"2025-04-30","arxiv_id":"2505.03788","repositories_listed":0,"syntology":null},{"url":null,"slug":"consens-assessing-context-grounding-in-open","title":"ConSens: Assessing context grounding in open-book question answering","date":"2025-04-30","arxiv_id":"2505.00065","repositories_listed":0,"syntology":null},{"url":null,"slug":"zoomer-adaptive-image-focus-optimization-for","title":"Zoomer: Adaptive Image Focus Optimization for Black-box MLLM","date":"2025-04-30","arxiv_id":"2505.00742","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-enhancer-merged-approach-using-vector","title":"LLM Enhancer: Merged Approach using Vector Embedding for Reducing Large Language Model Hallucinations with External Knowledge","date":"2025-04-29","arxiv_id":"2504.21132","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmme3dhf-benchmarking-and-evaluating","title":"LMME3DHF: Benchmarking and Evaluating Multimodal 3D Human Face Generation with LMMs","date":"2025-04-29","arxiv_id":"2504.20466","repositories_listed":0,"syntology":null},{"url":null,"slug":"setke-knowledge-editing-for-knowledge","title":"SetKE: Knowledge Editing for Knowledge Elements Overlap","date":"2025-04-29","arxiv_id":"2504.20972","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-of-domain-adapted-llms","title":"Knowledge Distillation of Domain-adapted LLMs for Question-Answering in Telecom","date":"2025-04-28","arxiv_id":"2504.20000","repositories_listed":0,"syntology":null},{"url":null,"slug":"m-kailin-knowledge-driven-agentic-scientific","title":"m-KAILIN: Knowledge-Driven Agentic Scientific Corpus Distillation Framework for Biomedical Large Language Models Training","date":"2025-04-28","arxiv_id":"2504.19565","repositories_listed":0,"syntology":null},{"url":null,"slug":"opentcm-a-graphrag-empowered-llm-based-system","title":"OpenTCM: A GraphRAG-Empowered LLM-based System for Traditional Chinese Medicine Knowledge Retrieval and Diagnosis","date":"2025-04-28","arxiv_id":"2504.20118","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatialreasoner-towards-explicit-and","title":"SpatialReasoner: Towards Explicit and Generalizable 3D Spatial Reasoning","date":"2025-04-28","arxiv_id":"2504.20024","repositories_listed":0,"syntology":null},{"url":null,"slug":"pushing-the-boundary-on-natural-language","title":"Pushing the boundary on Natural Language Inference","date":"2025-04-25","arxiv_id":"2504.18376","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-of-knowledge-based","title":"A Comprehensive Survey of Knowledge-Based Vision Question Answering Systems: The Lifecycle of Knowledge in Visual Reasoning Task","date":"2025-04-24","arxiv_id":"2504.17547","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-calibration-of-prediction-sets-in","title":"Data-Driven Calibration of Prediction Sets in Large Vision-Language Models Based on Inductive Conformal Prediction","date":"2025-04-24","arxiv_id":"2504.17671","repositories_listed":0,"syntology":null},{"url":null,"slug":"travellama-facilitating-multi-modal-large","title":"TraveLLaMA: Facilitating Multi-modal Large Language Models to Understand Urban Scenes and Provide Travel Assistance","date":"2025-04-23","arxiv_id":"2504.16505","repositories_listed":0,"syntology":null},{"url":null,"slug":"finder-financial-dataset-for-question","title":"FinDER: Financial Dataset for Question Answering and Evaluating Retrieval-Augmented Generation","date":"2025-04-22","arxiv_id":"2504.15800","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-great-nugget-recall-automating-fact","title":"The Great Nugget Recall: Automating Fact Extraction and RAG Evaluation with Large Language Models","date":"2025-04-21","arxiv_id":"2504.15068","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-camera-motions-in-any","title":"Towards Understanding Camera Motions in Any Video","date":"2025-04-21","arxiv_id":"2504.15376","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-framework-for-measuring","title":"A Hierarchical Framework for Measuring Scientific Paper Innovation via Large Language Models","date":"2025-04-20","arxiv_id":"2504.14620","repositories_listed":0,"syntology":null},{"url":null,"slug":"colota-a-dataset-for-entity-based-commonsense","title":"CoLoTa: A Dataset for Entity-based Commonsense Reasoning over Long-Tail Knowledge","date":"2025-04-20","arxiv_id":"2504.14462","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairsteer-inference-time-debiasing-for-llms","title":"FairSteer: Inference Time Debiasing for LLMs with Dynamic Activation Steering","date":"2025-04-20","arxiv_id":"2504.14492","repositories_listed":0,"syntology":null},{"url":null,"slug":"finsage-a-multi-aspect-rag-system-for","title":"FinSage: A Multi-aspect RAG System for Financial Filings Question Answering","date":"2025-04-20","arxiv_id":"2504.14493","repositories_listed":0,"syntology":null},{"url":null,"slug":"neglected-risks-the-disturbing-reality-of","title":"Neglected Risks: The Disturbing Reality of Children's Images in Datasets and the Urgent Call for Accountability","date":"2025-04-20","arxiv_id":"2504.14446","repositories_listed":0,"syntology":null},{"url":null,"slug":"bottom-up-synthesis-of-knowledge-grounded","title":"Bottom-Up Synthesis of Knowledge-Grounded Task-Oriented Dialogues with Iteratively Self-Refined Prompts","date":"2025-04-19","arxiv_id":"2504.14375","repositories_listed":0,"syntology":null},{"url":null,"slug":"legalrag-a-hybrid-rag-system-for-multilingual","title":"LegalRAG: A Hybrid RAG System for Multilingual Legal Information Retrieval","date":"2025-04-19","arxiv_id":"2504.16121","repositories_listed":0,"syntology":null},{"url":null,"slug":"sconu-selective-conformal-uncertainty-in","title":"SConU: Selective Conformal Uncertainty in Large Language Models","date":"2025-04-19","arxiv_id":"2504.14154","repositories_listed":0,"syntology":null},{"url":null,"slug":"accommodate-knowledge-conflicts-in-retrieval","title":"Accommodate Knowledge Conflicts in Retrieval-augmented LLMs: Towards Reliable Response Generation in the Wild","date":"2025-04-17","arxiv_id":"2504.12982","repositories_listed":0,"syntology":null},{"url":null,"slug":"chartqa-x-generating-explanations-for-charts","title":"ChartQA-X: Generating Explanations for Charts","date":"2025-04-17","arxiv_id":"2504.13275","repositories_listed":0,"syntology":null},{"url":null,"slug":"hadamard-product-in-deep-learning","title":"Hadamard product in deep learning: Introduction, Advances and Challenges","date":"2025-04-17","arxiv_id":"2504.13112","repositories_listed":0,"syntology":null},{"url":null,"slug":"weblists-extracting-structured-information","title":"WebLists: Extracting Structured Information From Complex Interactive Websites Using Executable LLM Agents","date":"2025-04-17","arxiv_id":"2504.12682","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-semantic-gaps-improving-medical","title":"Bridging the Semantic Gaps: Improving Medical VQA Consistency with LLM-Augmented Question Sets","date":"2025-04-16","arxiv_id":"2504.11777","repositories_listed":0,"syntology":null},{"url":null,"slug":"instruction-augmented-multimodal-alignment","title":"Instruction-augmented Multimodal Alignment for Image-Text and Element Matching","date":"2025-04-16","arxiv_id":"2504.12018","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-llm-hallucinations-with-knowledge","title":"Mitigating LLM Hallucinations with Knowledge Graphs: A Case Study","date":"2025-04-16","arxiv_id":"2504.12422","repositories_listed":0,"syntology":null},{"url":"/paper/self-alignment-of-large-video-language-models","slug":"self-alignment-of-large-video-language-models","title":"Self-alignment of Large Video Language Models with Refined Regularized Preference Optimization","date":"2025-04-16","arxiv_id":"2504.12083","repositories_listed":0,"syntology":null},{"url":null,"slug":"askqe-question-answering-as-automatic","title":"AskQE: Question Answering as Automatic Evaluation for Machine Translation","date":"2025-04-15","arxiv_id":"2504.11582","repositories_listed":0,"syntology":null},{"url":"/paper/benchmarking-biopharmaceuticals-retrieval","slug":"benchmarking-biopharmaceuticals-retrieval","title":"Benchmarking Biopharmaceuticals Retrieval-Augmented Generation Evaluation","date":"2025-04-15","arxiv_id":"2504.12342","repositories_listed":0,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/benchmarking-biopharmaceuticals-retrieval#ran","syntology_url":"https://syntology.ai/paper/2504.12342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.12342"}},"official":null}},{"url":null,"slug":"exploring-the-role-of-kg-based-rag-in","title":"Exploring the Role of Knowledge Graph-Based RAG in Japanese Medical Question Answering with Small-Scale LLMs","date":"2025-04-15","arxiv_id":"2504.10982","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-misleading-queries-to-accurate-answers-a","title":"From Misleading Queries to Accurate Answers: A Three-Stage Fine-Tuning Method for LLMs","date":"2025-04-15","arxiv_id":"2504.11277","repositories_listed":0,"syntology":null},{"url":null,"slug":"lvlm-csp-accelerating-large-vision-language","title":"LVLM_CSP: Accelerating Large Vision Language Models via Clustering, Scattering, and Pruning for Reasoning Segmentation","date":"2025-04-15","arxiv_id":"2504.10854","repositories_listed":0,"syntology":null},{"url":null,"slug":"streamlining-biomedical-research-with","title":"Streamlining Biomedical Research with Specialized LLMs","date":"2025-04-15","arxiv_id":"2504.12341","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-trustworthy-multimodal-ai-a-review","title":"Building Trustworthy Multimodal AI: A Review of Fairness, Transparency, and Ethics in Vision-Language Tasks","date":"2025-04-14","arxiv_id":"2504.13199","repositories_listed":0,"syntology":null},{"url":null,"slug":"constructing-micro-knowledge-graphs-from","title":"Constructing Micro Knowledge Graphs from Technical Support Documents","date":"2025-04-14","arxiv_id":"2504.09877","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-detection-in-llms-via","title":"Hallucination Detection in LLMs via Topological Divergence on Attention Graphs","date":"2025-04-14","arxiv_id":"2504.10063","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmkb-rag-a-multi-modal-knowledge-based","title":"MMKB-RAG: A Multi-Modal Knowledge-Based Retrieval-Augmented Generation Framework","date":"2025-04-14","arxiv_id":"2504.10074","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-court-combining-reasoning-action","title":"Reasoning Court: Combining Reasoning, Action, and Judgment for Multi-Hop Reasoning","date":"2025-04-14","arxiv_id":"2504.09781","repositories_listed":0,"syntology":null},{"url":null,"slug":"see-or-recall-a-sanity-check-for-the-role-of","title":"See or Recall: A Sanity Check for the Role of Vision in Solving Visualization Question Answer Tasks with Multimodal LLMs","date":"2025-04-14","arxiv_id":"2504.09809","repositories_listed":0,"syntology":null},{"url":null,"slug":"vdocrag-retrieval-augmented-generation-over","title":"VDocRAG: Retrieval-Augmented Generation over Visually-Rich Documents","date":"2025-04-14","arxiv_id":"2504.09795","repositories_listed":0,"syntology":null}],"record_sha256":"bbfe268cd3ff238d17d3fd5de74cc23f866b74838ad3c6f6337725f590d606ef","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}