{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/51","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":51,"pages_in_order":109,"rows_per_page":100,"rows":[5001,5100],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/50","next":"/task/question-answering/papers/52","papers":[{"url":null,"slug":"a-multimodal-social-agent","title":"A Multimodal Social Agent","date":"2024-12-11","arxiv_id":"2501.06189","repositories_listed":0,"syntology":null},{"url":null,"slug":"barking-up-the-syntactic-tree-enhancing-vlm","title":"Barking Up The Syntactic Tree: Enhancing VLM Training with Syntactic Losses","date":"2024-12-11","arxiv_id":"2412.08110","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-we-generate-visual-programs-without","title":"Can We Generate Visual Programs Without Prompting LLMs?","date":"2024-12-11","arxiv_id":"2412.08564","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialogagent-an-auto-engagement-agent-for-code","title":"DialogAgent: An Auto-engagement Agent for Code Question Answering Data Production","date":"2024-12-11","arxiv_id":"2412.08069","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-vision-language-tasks-benefit-from-large","title":"How Vision-Language Tasks Benefit from Large Pre-trained Models: A Survey","date":"2024-12-11","arxiv_id":"2412.08158","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-learning-with-topological","title":"In-Context Learning with Topological Information for Knowledge Graph Completion","date":"2024-12-11","arxiv_id":"2412.08742","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoprep-natural-language-question-aware-data","title":"AutoPrep: Natural Language Question-Aware Data Preparation with a Multi-Agent Framework","date":"2024-12-10","arxiv_id":"2412.10422","repositories_listed":0,"syntology":null},{"url":null,"slug":"ontology-aware-rag-for-improved-question","title":"Ontology-Aware RAG for Improved Question-Answering in Cybersecurity Education","date":"2024-12-10","arxiv_id":"2412.14191","repositories_listed":0,"syntology":null},{"url":"/paper/piece-of-table-a-divide-and-conquer-approach","slug":"piece-of-table-a-divide-and-conquer-approach","title":"Piece of Table: A Divide-and-Conquer Approach for Selecting Sub-Tables in Table Question Answering","date":"2024-12-10","arxiv_id":"2412.07629","repositories_listed":0,"syntology":null},{"url":"/paper/rag-based-question-answering-over","slug":"rag-based-question-answering-over","title":"RAG-based Question Answering over Heterogeneous Data and Text","date":"2024-12-10","arxiv_id":"2412.07420","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-spatial-understanding-in-mllms","title":"3D Spatial Understanding in MLLMs: Disambiguation and Evaluation","date":"2024-12-09","arxiv_id":"2412.06613","repositories_listed":0,"syntology":null},{"url":null,"slug":"ranked-from-within-ranking-large-multimodal","title":"Ranked from Within: Ranking Large Multimodal Models for Visual Question Answering Without Labels","date":"2024-12-09","arxiv_id":"2412.06461","repositories_listed":0,"syntology":null},{"url":null,"slug":"1-800-shared-tasks-at-regnlp-lexical","title":"1-800-SHARED-TASKS at RegNLP: Lexical Reranking of Semantic Retrieval (LeSeR) for Regulatory Question Answering","date":"2024-12-08","arxiv_id":"2412.06009","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-manufacturing-scale-up-from","title":"Accelerating Manufacturing Scale-Up from Material Discovery Using Agentic Web Navigation and Retrieval-Augmented AI for Process Engineering Schematics Design","date":"2024-12-08","arxiv_id":"2412.05937","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-entailment-tree-generation-approach-for","title":"An Entailment Tree Generation Approach for Multimodal Multi-Hop Question Answering with Mixture-of-Experts and Iterative Feedback Mechanism","date":"2024-12-08","arxiv_id":"2412.05821","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-hallucination-in-text-to-image","title":"Evaluating Hallucination in Text-to-Image Diffusion Models with Scene-Graph based Question-Answering Agent","date":"2024-12-07","arxiv_id":"2412.05722","repositories_listed":0,"syntology":null},{"url":null,"slug":"promptrefine-enhancing-few-shot-performance","title":"PromptRefine: Enhancing Few-Shot Performance on Low-Resource Indic Languages with Example Selection from Related Example Banks","date":"2024-12-07","arxiv_id":"2412.05710","repositories_listed":0,"syntology":null},{"url":null,"slug":"sla-management-in-reconfigurable-multi-agent","title":"SLA Management in Reconfigurable Multi-Agent RAG: A Systems Approach to Question Answering","date":"2024-12-07","arxiv_id":"2412.06832","repositories_listed":0,"syntology":null},{"url":null,"slug":"splaxbert-leveraging-mixed-precision-training","title":"SplaXBERT: Leveraging Mixed Precision Training and Context Splitting for Question Answering","date":"2024-12-07","arxiv_id":"2412.05499","repositories_listed":0,"syntology":null},{"url":null,"slug":"eaco-enhancing-alignment-in-multimodal-llms","title":"EACO: Enhancing Alignment in Multimodal LLMs via Critical Observation","date":"2024-12-06","arxiv_id":"2412.04903","repositories_listed":0,"syntology":null},{"url":null,"slug":"findings-of-the-second-babylm-challenge","title":"Findings of the Second BabyLM Challenge: Sample-Efficient Pretraining on Developmentally Plausible Corpora","date":"2024-12-06","arxiv_id":"2412.05149","repositories_listed":0,"syntology":null},{"url":null,"slug":"kalm-knowledge-aligned-autoregressive","title":"KaLM: Knowledge-aligned Autoregressive Language Modeling via Dual-view Knowledge Graph Contrastive Learning","date":"2024-12-06","arxiv_id":"2412.04948","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-graphs-are-all-you-need-leveraging","title":"Knowledge Graphs are all you need: Leveraging KGs in Physics Question Answering","date":"2024-12-06","arxiv_id":"2412.05453","repositories_listed":0,"syntology":null},{"url":null,"slug":"question-answering-for-decisionmaking-in","title":"Question Answering for Decisionmaking in Green Building Design: A Multimodal Data Reasoning Method Driven by Large Language Models","date":"2024-12-06","arxiv_id":"2412.04741","repositories_listed":0,"syntology":null},{"url":null,"slug":"steps-are-all-you-need-rethinking-stem","title":"Steps are all you need: Rethinking STEM Education with Prompt Engineering","date":"2024-12-06","arxiv_id":"2412.05023","repositories_listed":0,"syntology":null},{"url":null,"slug":"addressing-hallucinations-with-rag-and-nmiss","title":"Addressing Hallucinations with RAG and NMISS in Italian Healthcare LLM Chatbots","date":"2024-12-05","arxiv_id":"2412.04235","repositories_listed":0,"syntology":null},{"url":null,"slug":"comprehensive-audio-query-handling-system","title":"Comprehensive Audio Query Handling System with Integrated Expert Models and Contextual Understanding","date":"2024-12-05","arxiv_id":"2412.03980","repositories_listed":0,"syntology":null},{"url":null,"slug":"graf-graph-retrieval-augmented-by-facts-for","title":"GRAF: Graph Retrieval Augmented by Facts for Romanian Legal Multi-Choice Question Answering","date":"2024-12-05","arxiv_id":"2412.04119","repositories_listed":0,"syntology":null},{"url":null,"slug":"synergizing-llms-and-knowledge-graphs-a-novel","title":"Synergizing LLMs and Knowledge Graphs: A Novel Approach to Software Repository-Related Question Answering","date":"2024-12-05","arxiv_id":"2412.03815","repositories_listed":0,"syntology":null},{"url":null,"slug":"t2i-factualbench-benchmarking-the-factuality","title":"T2I-FactualBench: Benchmarking the Factuality of Text-to-Image Models with Knowledge-Intensive Concepts","date":"2024-12-05","arxiv_id":"2412.04300","repositories_listed":0,"syntology":null},{"url":null,"slug":"tango-training-free-embodied-ai-agents-for","title":"TANGO: Training-free Embodied AI Agents for Open-world Tasks","date":"2024-12-05","arxiv_id":"2412.10402","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniform-discretized-integrated-gradients-an","title":"Uniform Discretized Integrated Gradients: An effective attribution based method for explaining large language models","date":"2024-12-05","arxiv_id":"2412.03886","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-specific-question-answering-with","title":"Domain-specific Question Answering with Hybrid Search","date":"2024-12-04","arxiv_id":"2412.03736","repositories_listed":0,"syntology":null},{"url":null,"slug":"redstone-curating-general-code-math-and-qa","title":"RedStone: Curating General, Code, Math, and QA Data for Large Language Models","date":"2024-12-04","arxiv_id":"2412.03398","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-of-different-large-language-model","title":"Survey of different Large Language Model Architectures: Trends, Benchmarks, and Challenges","date":"2024-12-04","arxiv_id":"2412.03220","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-evolutionary-large-language-model-for","title":"An Evolutionary Large Language Model for Hallucination Mitigation","date":"2024-12-03","arxiv_id":"2412.02790","repositories_listed":0,"syntology":null},{"url":null,"slug":"cegi-measuring-the-trade-off-between","title":"CEGI: Measuring the trade-off between efficiency and carbon emissions for SLMs and VLMs","date":"2024-12-03","arxiv_id":"2412.02602","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-trust-in-large-language-models-with","title":"Enhancing Trust in Large Language Models with Uncertainty-Aware Fine-Tuning","date":"2024-12-03","arxiv_id":"2412.02904","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-and-interpretable-multimodal","title":"Explainable and Interpretable Multimodal Large Language Models: A Comprehensive Survey","date":"2024-12-03","arxiv_id":"2412.02104","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-squad-hybrid-scholarly-question","title":"Hybrid-SQuAD: Hybrid Scholarly Question Answering Dataset","date":"2024-12-03","arxiv_id":"2412.02788","repositories_listed":0,"syntology":null},{"url":null,"slug":"mld-ea-check-and-complete-narrative-coherence","title":"MLD-EA: Check and Complete Narrative Coherence by Introducing Emotions and Actions","date":"2024-12-03","arxiv_id":"2412.02897","repositories_listed":0,"syntology":null},{"url":null,"slug":"qa-toolbox-conversational-question-answering","title":"QA-TOOLBOX: Conversational Question-Answering for process task guidance in manufacturing","date":"2024-12-03","arxiv_id":"2412.02638","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-tokens-in-retrieval-augmented","title":"Semantic Tokens in Retrieval Augmented Generation","date":"2024-12-03","arxiv_id":"2412.02563","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignformer-modality-matching-can-achieve","title":"AlignFormer: Modality Matching Can Achieve Better Zero-shot Instruction-Following Speech-LLM","date":"2024-12-02","arxiv_id":"2412.01145","repositories_listed":0,"syntology":null},{"url":null,"slug":"mastering-board-games-by-external-and","title":"Mastering Board Games by External and Internal Planning with Language Models","date":"2024-12-02","arxiv_id":"2412.12119","repositories_listed":0,"syntology":null},{"url":null,"slug":"medchain-bridging-the-gap-between-llm-agents","title":"Medchain: Bridging the Gap Between LLM Agents and Clinical Practice through Interactive Sequential Benchmarking","date":"2024-12-02","arxiv_id":"2412.01605","repositories_listed":0,"syntology":null},{"url":null,"slug":"seal-semantic-attention-learning-for-long","title":"SEAL: Semantic Attention Learning for Long Video Representation","date":"2024-12-02","arxiv_id":"2412.01798","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-world-s-museums-through","title":"Understanding the World's Museums through Vision-Language Reasoning","date":"2024-12-02","arxiv_id":"2412.01370","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-video-llm-via-agent-of-thoughts","title":"Unlocking Video-LLM via Agent-of-Thoughts Distillation","date":"2024-12-02","arxiv_id":"2412.01694","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-language-models-potential-for","title":"Generative Language Models Potential for Requirement Engineering Applications: Insights into Current Strengths and Limitations","date":"2024-12-01","arxiv_id":"2412.00959","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-vietnamese-legal-document-retrieval","title":"Improving Vietnamese Legal Document Retrieval using Synthetic Data","date":"2024-12-01","arxiv_id":"2412.00657","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-unlearn-meta-learning-based","title":"Learn to Unlearn: Meta-Learning-Based Knowledge Graph Embedding Unlearning","date":"2024-12-01","arxiv_id":"2412.00881","repositories_listed":0,"syntology":null},{"url":null,"slug":"uhura-a-benchmark-for-evaluating-scientific","title":"Uhura: A Benchmark for Evaluating Scientific Question Answering and Truthfulness in Low-Resource African Languages","date":"2024-12-01","arxiv_id":"2412.00948","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynrank-improving-passage-retrieval-with","title":"DynRank: Improving Passage Retrieval with Dynamic Zero-Shot Prompting Based on Question Classification","date":"2024-11-30","arxiv_id":"2412.00600","repositories_listed":0,"syntology":null},{"url":null,"slug":"actions-and-objects-pathways-for-domain","title":"Actions and Objects Pathways for Domain Adaptation in Video Question Answering","date":"2024-11-29","arxiv_id":"2411.19434","repositories_listed":0,"syntology":null},{"url":null,"slug":"perception-test-2024-challenge-summary-and-a","title":"Perception Test 2024: Challenge Summary and a Novel Hour-Long VideoQA Benchmark","date":"2024-11-29","arxiv_id":"2411.19941","repositories_listed":0,"syntology":null},{"url":null,"slug":"step-enhancing-video-llms-compositional","title":"STEP: Enhancing Video-LLMs' Compositional Reasoning by Spatio-Temporal Graph-guided Self-Training","date":"2024-11-29","arxiv_id":"2412.00161","repositories_listed":0,"syntology":null},{"url":null,"slug":"unimib-assistant-designing-a-student-friendly","title":"Unimib Assistant: designing a student-friendly RAG-based chatbot for all their needs","date":"2024-11-29","arxiv_id":"2411.19554","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-logit-lens-contextual-embeddings-for","title":"Beyond Logit Lens: Contextual Embeddings for Robust Hallucination Detection & Grounding in VLMs","date":"2024-11-28","arxiv_id":"2411.19187","repositories_listed":0,"syntology":null},{"url":null,"slug":"diesel-dynamic-inference-guidance-via-evasion","title":"DIESEL -- Dynamic Inference-Guidance via Evasion of Semantic Embeddings in LLMs","date":"2024-11-28","arxiv_id":"2411.19038","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-attention-vectors-generative","title":"Sparse Attention Vectors: Generative Multimodal Model Features Are Discriminative Vision-Language Classifiers","date":"2024-11-28","arxiv_id":"2412.00142","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-scene-graph-guided-vision-language-pre","title":"3D Scene Graph Guided Vision-Language Pre-training","date":"2024-11-27","arxiv_id":"2411.18666","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-data-curation-effectively-distills","title":"Active Data Curation Effectively Distills Large-Scale Multimodal Models","date":"2024-11-27","arxiv_id":"2411.18674","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-bidirectional-encoder-become-the-ultimate","title":"Can bidirectional encoder become the ultimate winner for downstream applications of foundation models?","date":"2024-11-27","arxiv_id":"2411.18021","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-llms-assist-with-ambiguity-a-quantitative","title":"Can LLMs assist with Ambiguity? A Quantitative Evaluation of various Large Language Models on Word Sense Disambiguation","date":"2024-11-27","arxiv_id":"2411.18337","repositories_listed":0,"syntology":null},{"url":null,"slug":"electrovizqa-how-well-do-multi-modal-llms","title":"ElectroVizQA: How well do Multi-modal LLMs perform in Electronics Visual Question Answering?","date":"2024-11-27","arxiv_id":"2412.00102","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperglm-hypergraph-for-video-scene-graph","title":"HyperGLM: HyperGraph for Video Scene Graph Generation and Anticipation","date":"2024-11-27","arxiv_id":"2411.18042","repositories_listed":0,"syntology":null},{"url":null,"slug":"overview-of-trec-2024-biomedical-generative","title":"Overview of TREC 2024 Biomedical Generative Retrieval (BioGen) Track","date":"2024-11-27","arxiv_id":"2411.18069","repositories_listed":0,"syntology":null},{"url":null,"slug":"salmonn-omni-a-codec-free-llm-for-full-duplex","title":"SALMONN-omni: A Codec-free LLM for Full-duplex Speech Understanding and Generation","date":"2024-11-27","arxiv_id":"2411.18138","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-multi-modal-large-language-models","title":"Efficient Multi-modal Large Language Models via Visual Token Grouping","date":"2024-11-26","arxiv_id":"2411.17773","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-language-understanding-and-inference","title":"Natural Language Understanding and Inference with MLLM in Visual Question Answering: A Survey","date":"2024-11-26","arxiv_id":"2411.17558","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-progressive-curriculum-learning-for","title":"Task Progressive Curriculum Learning for Robust Visual Question Answering","date":"2024-11-26","arxiv_id":"2411.17292","repositories_listed":0,"syntology":null},{"url":null,"slug":"gemex-a-large-scale-groundable-and","title":"GEMeX: A Large-Scale, Groundable, and Explainable Medical VQA Benchmark for Chest X-ray Diagnosis","date":"2024-11-25","arxiv_id":"2411.16778","repositories_listed":0,"syntology":null},{"url":null,"slug":"videoorion-tokenizing-object-dynamics-in","title":"VideoOrion: Tokenizing Object Dynamics in Videos","date":"2024-11-25","arxiv_id":"2411.16156","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-guided-coarse-to-fine-fusion-network-for","title":"Text-Guided Coarse-to-Fine Fusion Network for Robust Remote Sensing Visual Question Answering","date":"2024-11-24","arxiv_id":"2411.15770","repositories_listed":0,"syntology":null},{"url":null,"slug":"afrimed-qa-a-pan-african-multi-specialty","title":"AfriMed-QA: A Pan-African, Multi-Specialty, Medical Question-Answering Benchmark Dataset","date":"2024-11-23","arxiv_id":"2411.15640","repositories_listed":0,"syntology":null},{"url":null,"slug":"finecaption-compositional-image-captioning","title":"FINECAPTION: Compositional Image Captioning Focusing on Wherever You Want at Any Granularity","date":"2024-11-23","arxiv_id":"2411.15411","repositories_listed":0,"syntology":null},{"url":null,"slug":"freepruner-a-training-free-approach-for-large","title":"freePruner: A Training-free Approach for Large Multimodal Model Acceleration","date":"2024-11-23","arxiv_id":"2411.15446","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewind-understanding-long-videos-with","title":"ReWind: Understanding Long Videos with Instructed Learnable Memory","date":"2024-11-23","arxiv_id":"2411.15556","repositories_listed":0,"syntology":null},{"url":null,"slug":"fastrag-retrieval-augmented-generation-for","title":"FastRAG: Retrieval Augmented Generation for Semi-structured Data","date":"2024-11-21","arxiv_id":"2411.13773","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-graphs-large-language-models-and","title":"Knowledge Graphs, Large Language Models, and Hallucinations: An NLP Perspective","date":"2024-11-21","arxiv_id":"2411.14258","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-llms-capabilities-towards","title":"Evaluating LLMs Capabilities Towards Understanding Social Dynamics","date":"2024-11-20","arxiv_id":"2411.13008","repositories_listed":0,"syntology":null},{"url":null,"slug":"lavida-drive-vision-text-interaction-vlm-for","title":"LaVida Drive: Vision-Text Interaction VLM for Autonomous Driving with Token Selection, Recovery and Enhancement","date":"2024-11-20","arxiv_id":"2411.12980","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reason-iteratively-and-parallelly","title":"Learning to Reason Iteratively and Parallelly for Complex Visual Reasoning Scenarios","date":"2024-11-20","arxiv_id":"2411.13754","repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieval-augmented-generation-for-domain-1","title":"Retrieval-Augmented Generation for Domain-Specific Question Answering: A Case Study on Pittsburgh and CMU","date":"2024-11-20","arxiv_id":"2411.13691","repositories_listed":0,"syntology":null},{"url":null,"slug":"uni-mlip-unified-self-supervision-for-medical","title":"Uni-Mlip: Unified Self-supervision for Medical Vision Language Pre-training","date":"2024-11-20","arxiv_id":"2411.15207","repositories_listed":0,"syntology":null},{"url":null,"slug":"adacm-2-on-understanding-extremely-long-term","title":"AdaCM$^2$: On Understanding Extremely Long-Term Video with Adaptive Cross-Modality Memory Reduction","date":"2024-11-19","arxiv_id":"2411.12593","repositories_listed":0,"syntology":null},{"url":null,"slug":"catch-complementary-adaptive-token-level","title":"CATCH: Complementary Adaptive Token-level Contrastive Decoding to Mitigate Hallucinations in LVLMs","date":"2024-11-19","arxiv_id":"2411.12713","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-llms-understand-ambiguity-in-text-a-case","title":"Do LLMs Understand Ambiguity in Text? A Case Study in Open-world Question Answering","date":"2024-11-19","arxiv_id":"2411.12395","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynfocus-dynamic-cooperative-network-empowers","title":"DynFocus: Dynamic Cooperative Network Empowers LLMs with Video Understanding","date":"2024-11-19","arxiv_id":"2411.12355","repositories_listed":0,"syntology":null},{"url":null,"slug":"med-2e3-a-2d-enhanced-3d-medical-multimodal","title":"Med-2E3: A 2D-Enhanced 3D Medical Multimodal Large Language Model","date":"2024-11-19","arxiv_id":"2411.12783","repositories_listed":0,"syntology":null},{"url":null,"slug":"textsc-neon-news-entity-interaction","title":"Neon: News Entity-Interaction Extraction for Enhanced Question Answering","date":"2024-11-19","arxiv_id":"2411.12449","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-knowledge-conflicts-in-language","title":"Mitigating Knowledge Conflicts in Language Model-Driven Question Answering","date":"2024-11-18","arxiv_id":"2411.11344","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-on-visual-question","title":"A Comprehensive Survey on Visual Question Answering Datasets and Algorithms","date":"2024-11-17","arxiv_id":"2411.11150","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-augmented-multimodal-llms-for-surgical","title":"Memory-Augmented Multimodal LLMs for Surgical VQA via Self-Contained Inquiry","date":"2024-11-17","arxiv_id":"2411.10937","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-multimodal-llms-the-mechanistic","title":"Understanding Multimodal LLMs: the Mechanistic Interpretability of Llava in Visual Question Answering","date":"2024-11-17","arxiv_id":"2411.10950","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-vision-language-models-for-remote","title":"Large Vision-Language Models for Remote Sensing Visual Question Answering","date":"2024-11-16","arxiv_id":"2411.10857","repositories_listed":0,"syntology":null},{"url":null,"slug":"llasa-large-language-and-structured-data","title":"LLaSA: Large Language and Structured Data Assistant","date":"2024-11-16","arxiv_id":"2411.14460","repositories_listed":0,"syntology":null},{"url":null,"slug":"amxfp4-taming-activation-outliers-with","title":"AMXFP4: Taming Activation Outliers with Asymmetric Microscaling Floating-Point for 4-bit LLM Inference","date":"2024-11-15","arxiv_id":"2411.09909","repositories_listed":0,"syntology":null},{"url":null,"slug":"everything-is-a-video-unifying-modalities","title":"Everything is a Video: Unifying Modalities through Next-Frame Prediction","date":"2024-11-15","arxiv_id":"2411.10503","repositories_listed":0,"syntology":null}],"record_sha256":"154674ee55f639dc152940a4b95a4464b75eaf46e5d1c96a17bdbe76a903566e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}