{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/57","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":57,"pages_in_order":109,"rows_per_page":100,"rows":[5601,5700],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/56","next":"/task/question-answering/papers/58","papers":[{"url":null,"slug":"zero-shot-long-form-video-understanding","title":"Zero-Shot Long-Form Video Understanding through Screenplay","date":"2024-06-25","arxiv_id":"2406.17309","repositories_listed":0,"syntology":null},{"url":null,"slug":"advscore-a-metric-for-the-evaluation-and","title":"Is your benchmark truly adversarial? AdvScore: Evaluating Human-Grounded Adversarialness","date":"2024-06-24","arxiv_id":"2406.16342","repositories_listed":0,"syntology":null},{"url":"/paper/claude-3-5-sonnet-model-card-addendum","slug":"claude-3-5-sonnet-model-card-addendum","title":"Claude 3.5 Sonnet Model Card Addendum","date":"2024-06-24","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"context-augmented-retrieval-a-novel-framework","title":"Context-augmented Retrieval: A Novel Framework for Fast Information Retrieval based Response Generation using Large Language Model","date":"2024-06-24","arxiv_id":"2406.16383","repositories_listed":0,"syntology":null},{"url":null,"slug":"directed-domain-fine-tuning-tailoring","title":"Directed Domain Fine-Tuning: Tailoring Separate Modalities for Specific Training Tasks","date":"2024-06-24","arxiv_id":"2406.16346","repositories_listed":0,"syntology":null},{"url":null,"slug":"gpt-4v-explorations-mining-autonomous-driving","title":"GPT-4V Explorations: Mining Autonomous Driving","date":"2024-06-24","arxiv_id":"2406.16817","repositories_listed":0,"syntology":null},{"url":null,"slug":"mm-spubench-towards-better-understanding-of","title":"MM-SpuBench: Towards Better Understanding of Spurious Biases in Multimodal LLMs","date":"2024-06-24","arxiv_id":"2406.17126","repositories_listed":0,"syntology":null},{"url":null,"slug":"modulating-language-model-experiences-through","title":"Modulating Language Model Experiences through Frictions","date":"2024-06-24","arxiv_id":"2407.12804","repositories_listed":0,"syntology":null},{"url":null,"slug":"seam-a-stochastic-benchmark-for-multi","title":"SEAM: A Stochastic Benchmark for Multi-Document Tasks","date":"2024-06-23","arxiv_id":"2406.16086","repositories_listed":0,"syntology":null},{"url":null,"slug":"mr-mllm-mutual-reinforcement-of-multimodal","title":"MR-MLLM: Mutual Reinforcement of Multimodal Comprehension and Vision Perception","date":"2024-06-22","arxiv_id":"2406.15768","repositories_listed":0,"syntology":null},{"url":null,"slug":"70b-parameter-large-language-models-in","title":"70B-parameter large language models in Japanese medical question-answering","date":"2024-06-21","arxiv_id":"2406.14882","repositories_listed":0,"syntology":null},{"url":null,"slug":"generate-then-ground-in-retrieval-augmented","title":"Generate-then-Ground in Retrieval-Augmented Generation for Multi-hop Question Answering","date":"2024-06-21","arxiv_id":"2406.14891","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-whisper-for-qa-driven-zero-shot-end","title":"Prompting Whisper for QA-driven Zero-shot End-to-end Spoken Language Understanding","date":"2024-06-21","arxiv_id":"2406.15209","repositories_listed":0,"syntology":null},{"url":null,"slug":"sports-intelligence-assessing-the-sports","title":"Sports Intelligence: Assessing the Sports Understanding Capabilities of Language Models through Question Answering from Text to Video","date":"2024-06-21","arxiv_id":"2406.14877","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-retrieval-augmented-generation-over","title":"Towards Retrieval Augmented Generation over Large Video Libraries","date":"2024-06-21","arxiv_id":"2406.14938","repositories_listed":0,"syntology":null},{"url":null,"slug":"tri-vqa-triangular-reasoning-medical-visual","title":"Tri-VQA: Triangular Reasoning Medical Visual Question Answering for Multi-Attribute Analysis","date":"2024-06-21","arxiv_id":"2406.15050","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-learn-then-reason-model-towards","title":"A Learn-Then-Reason Model Towards Generalization in Knowledge Base Question Answering","date":"2024-06-20","arxiv_id":"2406.14763","repositories_listed":0,"syntology":null},{"url":null,"slug":"does-object-grounding-really-reduce","title":"Does Object Grounding Really Reduce Hallucination of Large Vision-Language Models?","date":"2024-06-20","arxiv_id":"2406.14492","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-mysteries-of-cot-augmented","title":"Investigating Mysteries of CoT-Augmented Distillation","date":"2024-06-20","arxiv_id":"2406.14511","repositories_listed":0,"syntology":null},{"url":null,"slug":"pku-saferlhf-a-safety-alignment-preference","title":"PKU-SafeRLHF: Towards Multi-Level Safety Alignment for LLMs with Human Preference","date":"2024-06-20","arxiv_id":"2406.15513","repositories_listed":0,"syntology":null},{"url":null,"slug":"ranking-llms-by-compression","title":"Ranking LLMs by compression","date":"2024-06-20","arxiv_id":"2406.14171","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-few-shot-transfer-learning-for","title":"Robust Few-shot Transfer Learning for Knowledge Base Question Answering with Unanswerable Questions","date":"2024-06-20","arxiv_id":"2406.14313","repositories_listed":0,"syntology":null},{"url":null,"slug":"syndarin-synthesising-datasets-for-automated","title":"SynDARin: Synthesising Datasets for Automated Reasoning in Low-Resource Languages","date":"2024-06-20","arxiv_id":"2406.14425","repositories_listed":0,"syntology":null},{"url":null,"slug":"temporal-knowledge-graph-question-answering-a","title":"Temporal Knowledge Graph Question Answering: A Survey","date":"2024-06-20","arxiv_id":"2406.14191","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-fire-thief-is-also-the-keeper-balancing","title":"The Fire Thief Is Also the Keeper: Balancing Usability and Privacy in Prompts","date":"2024-06-20","arxiv_id":"2406.14318","repositories_listed":0,"syntology":null},{"url":null,"slug":"ttqa-rs-a-break-down-prompting-approach-for","title":"TTQA-RS- A break-down prompting approach for Multi-hop Table-Text Question Answering with Reasoning and Summarization","date":"2024-06-20","arxiv_id":"2406.14732","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-finetuning-for-factual-1","title":"Understanding Finetuning for Factual Knowledge Extraction","date":"2024-06-20","arxiv_id":"2406.14785","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-open-source-language-models-for","title":"Comparison of Open-Source and Proprietary LLMs for Machine Reading Comprehension: A Practical Analysis for Industrial Applications","date":"2024-06-19","arxiv_id":"2406.13713","repositories_listed":0,"syntology":null},{"url":null,"slug":"forag-factuality-optimized-retrieval","title":"FoRAG: Factuality-optimized Retrieval Augmented Generation for Web-enhanced Long-form Question Answering","date":"2024-06-19","arxiv_id":"2406.13779","repositories_listed":0,"syntology":null},{"url":null,"slug":"qrmem-unleash-the-length-limitation-through","title":"QRMeM: Unleash the Length Limitation through Question then Reflection Memory Mechanism","date":"2024-06-19","arxiv_id":"2406.13167","repositories_listed":0,"syntology":null},{"url":null,"slug":"thread-a-logic-based-data-organization","title":"Thread: A Logic-Based Data Organization Paradigm for How-To Question Answering with Retrieval Augmented Generation","date":"2024-06-19","arxiv_id":"2406.13372","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-evaluation-a-comprehensive","title":"Towards Robust Evaluation: A Comprehensive Taxonomy of Datasets and Metrics for Open Domain Question Answering in the Era of Large Language Models","date":"2024-06-19","arxiv_id":"2406.13232","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-speech-to-text-large-language","title":"Transferable speech-to-text large language model alignment module","date":"2024-06-19","arxiv_id":"2406.13357","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-compass-for-navigating-the-world-of","title":"Towards Understanding Domain Adapted Sentence Embeddings for Document Retrieval","date":"2024-06-18","arxiv_id":"2406.12336","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-rags-to-rich-parameters-probing-how","title":"From RAGs to rich parameters: Probing how language models utilize external knowledge over parametric information for factual queries","date":"2024-06-18","arxiv_id":"2406.12824","repositories_listed":0,"syntology":null},{"url":null,"slug":"intermediate-distillation-data-efficient","title":"Intermediate Distillation: Data-Efficient Distillation from Black-Box LLMs for Information Retrieval","date":"2024-06-18","arxiv_id":"2406.12169","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightpal-lightweight-passage-retrieval-for","title":"LightPAL: Lightweight Passage Retrieval for Open Domain Multi-Document Summarization","date":"2024-06-18","arxiv_id":"2406.12494","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-robustness-of-language-models-for","title":"Exploring the Robustness of Language Models for Tabular Question Answering via Attention Analysis","date":"2024-06-18","arxiv_id":"2406.12719","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-not-design-learn-a-trainable-scoring","title":"Do Not Design, Learn: A Trainable Scoring Function for Uncertainty Estimation in Generative LLMs","date":"2024-06-17","arxiv_id":"2406.11278","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-biomedical-knowledge-retrieval","title":"SeRTS: Self-Rewarding Tree Search for Biomedical Retrieval-Augmented Generation","date":"2024-06-17","arxiv_id":"2406.11258","repositories_listed":0,"syntology":null},{"url":null,"slug":"hallucination-mitigation-prompts-long-term","title":"Hallucination Mitigation Prompts Long-term Video Understanding","date":"2024-06-17","arxiv_id":"2406.11333","repositories_listed":0,"syntology":null},{"url":null,"slug":"internalinspector-i-2-robust-confidence","title":"InternalInspector $I^2$: Robust Confidence Estimation in LLMs through Internal States","date":"2024-06-17","arxiv_id":"2406.12053","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-utility-judgment-framework-via-llms","title":"Iterative Utility Judgment Framework via LLMs Inspired by Relevance in Philosophy","date":"2024-06-17","arxiv_id":"2406.11290","repositories_listed":0,"syntology":null},{"url":null,"slug":"llarva-vision-action-instruction-tuning","title":"LLARVA: Vision-Action Instruction Tuning Enhances Robot Learning","date":"2024-06-17","arxiv_id":"2406.11815","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-large-language-model-hallucination","title":"Mitigating Large Language Model Hallucination with Faithful Finetuning","date":"2024-06-17","arxiv_id":"2406.11267","repositories_listed":0,"syntology":null},{"url":null,"slug":"move-beyond-triples-contextual-knowledge","title":"Context Graph","date":"2024-06-17","arxiv_id":"2406.11160","repositories_listed":0,"syntology":null},{"url":null,"slug":"program-synthesis-benchmark-for-visual","title":"Program Synthesis Benchmark for Visual Programming in XLogoOnline Environment","date":"2024-06-17","arxiv_id":"2406.11334","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-query-rewriting-aligning-rewriters","title":"Adaptive Query Rewriting: Aligning Rewriters through Marginal Probability of Conversational Answers","date":"2024-06-16","arxiv_id":"2406.10991","repositories_listed":0,"syntology":null},{"url":null,"slug":"embodied-question-answering-via-multi-llm","title":"Multi-LLM QA with Embodied Exploration","date":"2024-06-16","arxiv_id":"2406.10918","repositories_listed":0,"syntology":null},{"url":null,"slug":"hiddentables-pyqtax-a-cooperative-game-and","title":"HiddenTables & PyQTax: A Cooperative Game and Dataset For TableQA to Ensure Scale and Data Privacy Across a Myriad of Taxonomies","date":"2024-06-16","arxiv_id":"2406.10803","repositories_listed":0,"syntology":null},{"url":null,"slug":"theanine-revisiting-memory-management-in-long","title":"Towards Lifelong Dialogue Agents via Timeline-based Memory Management","date":"2024-06-16","arxiv_id":"2406.10996","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-hardness-of-faithful-chain-of-thought","title":"On the Hardness of Faithful Chain-of-Thought Reasoning in Large Language Models","date":"2024-06-15","arxiv_id":"2406.10625","repositories_listed":0,"syntology":null},{"url":null,"slug":"reasoning-or-simply-next-token-prediction-a","title":"MMLU-SR: A Benchmark for Stress-Testing Reasoning Capability of Large Language Models","date":"2024-06-15","arxiv_id":"2406.15468","repositories_listed":0,"syntology":null},{"url":null,"slug":"vceval-rethinking-what-is-a-good-educational","title":"VCEval: Rethinking What is a Good Educational Video and How to Automatically Evaluate It","date":"2024-06-15","arxiv_id":"2407.12005","repositories_listed":0,"syntology":null},{"url":null,"slug":"datasets-for-multilingual-answer-sentence","title":"Datasets for Multilingual Answer Sentence Selection","date":"2024-06-14","arxiv_id":"2406.10172","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-and-evaluating-medical","title":"Detecting and Evaluating Medical Hallucinations in Large Vision Language Models","date":"2024-06-14","arxiv_id":"2406.10185","repositories_listed":0,"syntology":null},{"url":"/paper/efficient-prompting-for-llm-based-generative","slug":"efficient-prompting-for-llm-based-generative","title":"Efficient Prompting for LLM-based Generative Internet of Things","date":"2024-06-14","arxiv_id":"2406.10382","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-question-answering-on-charts","title":"Enhancing Question Answering on Charts Through Effective Pre-training Tasks","date":"2024-06-14","arxiv_id":"2406.10085","repositories_listed":0,"syntology":null},{"url":null,"slug":"ewek-qa-enhanced-web-and-efficient-knowledge","title":"EWEK-QA: Enhanced Web and Efficient Knowledge Graph Retrieval for Citation-based Question Answering Systems","date":"2024-06-14","arxiv_id":"2406.10393","repositories_listed":0,"syntology":null},{"url":null,"slug":"gliner-multi-task-generalist-lightweight","title":"GLiNER multi-task: Generalist Lightweight Model for Various Information Extraction Tasks","date":"2024-06-14","arxiv_id":"2406.12925","repositories_listed":0,"syntology":null},{"url":null,"slug":"hip-attention-sparse-sub-quadratic-attention","title":"A Training-free Sub-quadratic Cost Transformer Model Serving Framework With Hierarchically Pruned Attention","date":"2024-06-14","arxiv_id":"2406.09827","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-large-language-models-with-graph","title":"Integrating Large Language Models with Graph-based Reasoning for Conversational Question Answering","date":"2024-06-14","arxiv_id":"2407.09506","repositories_listed":0,"syntology":null},{"url":null,"slug":"precision-empowers-excess-distracts-visual","title":"Precision Empowers, Excess Distracts: Visual Question Answering With Dynamically Infused Knowledge In Language Models","date":"2024-06-14","arxiv_id":"2406.09994","repositories_listed":0,"syntology":null},{"url":null,"slug":"shmamba-structured-hyperbolic-state-space","title":"SHMamba: Structured Hyperbolic State Space Model for Audio-Visual Question Answering","date":"2024-06-14","arxiv_id":"2406.09833","repositories_listed":0,"syntology":null},{"url":null,"slug":"discreteslu-a-large-language-model-with-self","title":"DiscreteSLU: A Large Language Model with Self-Supervised Discrete Speech Units for Spoken Language Understanding","date":"2024-06-13","arxiv_id":"2406.09345","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-modal-retrieval-for-large-language","title":"Multi-Modal Retrieval For Large Language Model Based Speech Recognition","date":"2024-06-13","arxiv_id":"2406.09618","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-visual-question-answering-models","title":"Optimizing Visual Question Answering Models for Driving: Bridging the Gap Between Human and Machine Attention Patterns","date":"2024-06-13","arxiv_id":"2406.09203","repositories_listed":0,"syntology":null},{"url":null,"slug":"distildoc-knowledge-distillation-for-visually","title":"DistilDoc: Knowledge Distillation for Visually-Rich Document Applications","date":"2024-06-12","arxiv_id":"2406.08226","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-stochastic-decoding-strategy-for-open","title":"Dynamic Stochastic Decoding Strategy for Open-Domain Dialogue Generation","date":"2024-06-12","arxiv_id":"2406.07850","repositories_listed":0,"syntology":null},{"url":null,"slug":"prediction-of-the-realisation-of-an","title":"Prediction of the Realisation of an Information Need: An EEG Study","date":"2024-06-12","arxiv_id":"2406.08105","repositories_listed":0,"syntology":null},{"url":null,"slug":"research-trends-for-the-interplay-between","title":"Research Trends for the Interplay between Large Language Models and Knowledge Graphs","date":"2024-06-12","arxiv_id":"2406.08223","repositories_listed":0,"syntology":null},{"url":null,"slug":"dr-rag-applying-dynamic-document-relevance-to","title":"DR-RAG: Applying Dynamic Document Relevance to Retrieval-Augmented Generation for Question-Answering","date":"2024-06-11","arxiv_id":"2406.07348","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-parallel-multi-hop-reasoning-a","title":"Efficient Parallel Multi-Hop Reasoning: A Scalable Approach for Knowledge Graph Analysis","date":"2024-06-11","arxiv_id":"2406.07727","repositories_listed":0,"syntology":null},{"url":null,"slug":"paraphrasing-in-affirmative-terms-improves","title":"Paraphrasing in Affirmative Terms Improves Negation Understanding","date":"2024-06-11","arxiv_id":"2406.07492","repositories_listed":0,"syntology":null},{"url":null,"slug":"question-answering-qa-model-for-a","title":"Question-Answering (QA) Model for a Personalized Learning Assistant for Arabic Language","date":"2024-06-11","arxiv_id":"2406.08519","repositories_listed":0,"syntology":null},{"url":null,"slug":"brainchat-decoding-semantic-information-from","title":"BrainChat: Decoding Semantic Information from fMRI using Vision-language Pretrained Models","date":"2024-06-10","arxiv_id":"2406.07584","repositories_listed":0,"syntology":null},{"url":null,"slug":"cvqa-culturally-diverse-multilingual-visual","title":"CVQA: Culturally-diverse Multilingual Visual Question Answering Benchmark","date":"2024-06-10","arxiv_id":"2406.05967","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-the-retrieval-component-in-llm","title":"Evaluating the Retrieval Component in LLM-Based Question Answering Systems","date":"2024-06-10","arxiv_id":"2406.06458","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-ai-for-efficient-analysis-of","title":"Harnessing AI for efficient analysis of complex policy documents: a case study of Executive Order 14110","date":"2024-06-10","arxiv_id":"2406.06657","repositories_listed":0,"syntology":null},{"url":null,"slug":"holmes-hyper-relational-knowledge-graphs-for","title":"HOLMES: Hyper-Relational Knowledge Graphs for Multi-hop Question Answering using LLMs","date":"2024-06-10","arxiv_id":"2406.06027","repositories_listed":0,"syntology":null},{"url":null,"slug":"solution-for-smart-101-challenge-of-cvpr","title":"Solution for SMART-101 Challenge of CVPR Multi-modal Algorithmic Reasoning Task 2024","date":"2024-06-10","arxiv_id":"2406.05963","repositories_listed":0,"syntology":null},{"url":null,"slug":"transforming-wearable-data-into-health","title":"Transforming Wearable Data into Health Insights using Large Language Model Agents","date":"2024-06-10","arxiv_id":"2406.06464","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-llms-exhibit-human-like-reasoning","title":"Do LLMs Exhibit Human-Like Reasoning? Evaluating Theory of Mind in LLMs for Open-Ended Responses","date":"2024-06-09","arxiv_id":"2406.05659","repositories_listed":0,"syntology":null},{"url":null,"slug":"medreqal-examining-medical-knowledge-recall","title":"MedREQAL: Examining Medical Knowledge Recall of Large Language Models via Question Answering","date":"2024-06-09","arxiv_id":"2406.05845","repositories_listed":0,"syntology":null},{"url":null,"slug":"mrrank-improving-question-answering-retrieval","title":"MrRank: Improving Question Answering Retrieval System through Multi-Result Ranking Model","date":"2024-06-09","arxiv_id":"2406.05733","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-end-to-end-spoken-question","title":"Zero-Shot End-To-End Spoken Question Answering In Medical Domain","date":"2024-06-09","arxiv_id":"2406.05876","repositories_listed":0,"syntology":null},{"url":null,"slug":"calm-contrasting-large-and-small-language","title":"CaLM: Contrasting Large and Small Language Models to Verify Grounded Generation","date":"2024-06-08","arxiv_id":"2406.05365","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-llms-recognize-me-when-i-is-not-me","title":"Do LLMs Recognize me, When I is not me: Assessment of LLMs Understanding of Turkish Indexical Pronouns in Indexical Shift Contexts","date":"2024-06-08","arxiv_id":"2406.05569","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-and-addressing-hallucinations","title":"Investigating and Addressing Hallucinations of LLMs in Tasks Involving Negation","date":"2024-06-08","arxiv_id":"2406.05494","repositories_listed":0,"syntology":null},{"url":null,"slug":"venn-diagram-prompting-accelerating","title":"Venn Diagram Prompting : Accelerating Comprehension with Scaffolding Effect","date":"2024-06-08","arxiv_id":"2406.05369","repositories_listed":0,"syntology":null},{"url":null,"slug":"criskeval-a-chinese-multi-level-risk","title":"CRiskEval: A Chinese Multi-Level Risk Evaluation Benchmark Dataset for Large Language Models","date":"2024-06-07","arxiv_id":"2406.04752","repositories_listed":0,"syntology":null},{"url":null,"slug":"matter-memory-augmented-transformer-using","title":"MATTER: Memory-Augmented Transformer Using Heterogeneous Knowledge Sources","date":"2024-06-07","arxiv_id":"2406.04670","repositories_listed":0,"syntology":null},{"url":null,"slug":"tcmd-a-traditional-chinese-medicine-qa","title":"TCMD: A Traditional Chinese Medicine QA Dataset for Evaluating Large Language Models","date":"2024-06-07","arxiv_id":"2406.04941","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-knowledge-infusion-via-kg-llm","title":"Efficient Knowledge Infusion via KG-LLM Alignment","date":"2024-06-06","arxiv_id":"2406.03746","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesizing-conversations-from-unlabeled","title":"Synthesizing Conversations from Unlabeled Documents using Automatic Response Segmentation","date":"2024-06-06","arxiv_id":"2406.03703","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-information-storage-and","title":"Understanding Information Storage and Transfer in Multi-modal Large Language Models","date":"2024-06-06","arxiv_id":"2406.04236","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-has-predicting-downstream-capabilities-of","title":"Why Has Predicting Downstream Capabilities of Frontier AI Models with Scale Remained Elusive?","date":"2024-06-06","arxiv_id":"2406.04391","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-performance-and-efficiency-in-zero","title":"Balancing Performance and Efficiency in Zero-shot Robotic Navigation","date":"2024-06-05","arxiv_id":"2406.03015","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-bias-in-latent-space-an","title":"Discovering Bias in Latent Space: An Unsupervised Debiasing Approach","date":"2024-06-05","arxiv_id":"2406.03631","repositories_listed":0,"syntology":null},{"url":null,"slug":"irokobench-a-new-benchmark-for-african","title":"IrokoBench: A New Benchmark for African Languages in the Age of Large Language Models","date":"2024-06-05","arxiv_id":"2406.03368","repositories_listed":0,"syntology":null}],"record_sha256":"f50680e9be3614174db835a2948410cb89c47e66b800b025b309eb085ab56d25","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}