{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/question-answering/papers/10","list_of":"/task/question-answering","task":"Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":109,"rows_per_page":100,"rows":[901,1000],"of":10817,"counts":{"archive_papers_tagged":10817,"with_a_code_link":4171,"where_syntology_ran_a_sample":1274,"not_listed_spam_title":0,"listed":10817,"listed_where_code_ran":1274,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1073,"every_run_a_failure_of_syntologys_instrument":201,"listed_with_a_run_with_no_instrument_failure":1073,"listed_every_run_a_failure_of_syntologys_instrument":201,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/question-answering","prev":"/task/question-answering/papers/9","next":"/task/question-answering/papers/11","papers":[{"url":"/paper/a-survey-on-efficient-vision-language-models","slug":"a-survey-on-efficient-vision-language-models","title":"A Survey on Efficient Vision-Language Models","date":"2025-04-13","arxiv_id":"2504.09724","repositories_listed":1,"syntology":null},{"url":"/paper/tinyllava-video-r1-towards-smaller-lmms-for","slug":"tinyllava-video-r1-towards-smaller-lmms-for","title":"TinyLLaVA-Video-R1: Towards Smaller LMMs for Video Reasoning","date":"2025-04-13","arxiv_id":"2504.09641","repositories_listed":1,"syntology":null},{"url":"/paper/out-of-style-rag-s-fragility-to-linguistic","slug":"out-of-style-rag-s-fragility-to-linguistic","title":"Out of Style: RAG's Fragility to Linguistic Variation","date":"2025-04-11","arxiv_id":"2504.08231","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/out-of-style-rag-s-fragility-to-linguistic#ran","syntology_url":"https://syntology.ai/paper/2504.08231","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.08231"}},"official":{"repos":["springcty/rag-fragility-to-linguistic-variation"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rag-vr-leveraging-retrieval-augmented","slug":"rag-vr-leveraging-retrieval-augmented","title":"RAG-VR: Leveraging Retrieval-Augmented Generation for 3D Question Answering in VR Environments","date":"2025-04-11","arxiv_id":"2504.08256","repositories_listed":1,"syntology":null},{"url":"/paper/do-llms-understand-your-translations","slug":"do-llms-understand-your-translations","title":"Do LLMs Understand Your Translations? Evaluating Paragraph-level MT with Question Answering","date":"2025-04-10","arxiv_id":"2504.07583","repositories_listed":1,"syntology":null},{"url":"/paper/mrd-rag-enhancing-medical-diagnosis-with","slug":"mrd-rag-enhancing-medical-diagnosis-with","title":"MRD-RAG: Enhancing Medical Diagnosis with Multi-Round Retrieval-Augmented Generation","date":"2025-04-10","arxiv_id":"2504.07724","repositories_listed":1,"syntology":null},{"url":"/paper/plan-and-refine-diverse-and-comprehensive","slug":"plan-and-refine-diverse-and-comprehensive","title":"Plan-and-Refine: Diverse and Comprehensive Retrieval-Augmented Generation","date":"2025-04-10","arxiv_id":"2504.07794","repositories_listed":1,"syntology":null},{"url":"/paper/resource-efficient-inference-with-foundation","slug":"resource-efficient-inference-with-foundation","title":"Resource-efficient Inference with Foundation Model Programs","date":"2025-04-09","arxiv_id":"2504.07247","repositories_listed":1,"syntology":null},{"url":"/paper/towards-an-ai-driven-video-based-american","slug":"towards-an-ai-driven-video-based-american","title":"Towards an AI-Driven Video-Based American Sign Language Dictionary: Exploring Design and Usage Experience with Learners","date":"2025-04-08","arxiv_id":"2504.05857","repositories_listed":1,"syntology":null},{"url":"/paper/chartqapro-a-more-diverse-and-challenging","slug":"chartqapro-a-more-diverse-and-challenging","title":"ChartQAPro: A More Diverse and Challenging Benchmark for Chart Question Answering","date":"2025-04-07","arxiv_id":"2504.05506","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chartqapro-a-more-diverse-and-challenging#ran","syntology_url":"https://syntology.ai/paper/2504.05506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05506"}},"official":{"repos":["vis-nlp/chartqapro"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/collab-rag-boosting-retrieval-augmented","slug":"collab-rag-boosting-retrieval-augmented","title":"Collab-RAG: Boosting Retrieval-Augmented Generation for Complex Question Answering via White-Box and Black-Box LLM Collaboration","date":"2025-04-07","arxiv_id":"2504.04915","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/collab-rag-boosting-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2504.04915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.04915"}},"official":{"repos":["ritaranx/collab-rag"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/enhancing-compositional-reasoning-in-vision","slug":"enhancing-compositional-reasoning-in-vision","title":"Enhancing Compositional Reasoning in Vision-Language Models with Synthetic Preference Data","date":"2025-04-07","arxiv_id":"2504.04740","repositories_listed":1,"syntology":null},{"url":"/paper/arxivbench-can-llms-assist-researchers-in","slug":"arxivbench-can-llms-assist-researchers-in","title":"ArxivBench: Can LLMs Assist Researchers in Conducting Research?","date":"2025-04-06","arxiv_id":"2504.10496","repositories_listed":1,"syntology":null},{"url":"/paper/medm-vl-what-makes-a-good-medical-lvlm","slug":"medm-vl-what-makes-a-good-medical-lvlm","title":"MedM-VL: What Makes a Good Medical LVLM?","date":"2025-04-06","arxiv_id":"2504.04323","repositories_listed":1,"syntology":null},{"url":"/paper/sigma-a-dataset-for-text-to-code-semantic","slug":"sigma-a-dataset-for-text-to-code-semantic","title":"Sigma: A dataset for text-to-code semantic parsing with statistical analysis","date":"2025-04-05","arxiv_id":"2504.04301","repositories_listed":1,"syntology":null},{"url":"/paper/generative-ai-enhanced-financial-risk","slug":"generative-ai-enhanced-financial-risk","title":"Generative AI Enhanced Financial Risk Management Information Retrieval","date":"2025-04-04","arxiv_id":"2504.06293","repositories_listed":1,"syntology":null},{"url":"/paper/sarlang-1m-a-benchmark-for-vision-language","slug":"sarlang-1m-a-benchmark-for-vision-language","title":"SARLANG-1M: A Benchmark for Vision-Language Modeling in SAR Image Understanding","date":"2025-04-04","arxiv_id":"2504.03254","repositories_listed":1,"syntology":null},{"url":"/paper/single-pass-document-scanning-for-question","slug":"single-pass-document-scanning-for-question","title":"Single-Pass Document Scanning for Question Answering","date":"2025-04-04","arxiv_id":"2504.03101","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/single-pass-document-scanning-for-question#ran","syntology_url":"https://syntology.ai/paper/2504.03101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.03101"}},"official":{"repos":["mambaretriever/mambaretriever"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/yalenlp-peranssumm-2025-multi-perspective-1","slug":"yalenlp-peranssumm-2025-multi-perspective-1","title":"YaleNLP @ PerAnsSumm 2025: Multi-Perspective Integration via Mixture-of-Agents for Enhanced Healthcare QA Summarization","date":"2025-04-04","arxiv_id":"2504.03932","repositories_listed":1,"syntology":null},{"url":"/paper/sting-bee-towards-vision-language-model-for","slug":"sting-bee-towards-vision-language-model-for","title":"STING-BEE: Towards Vision-Language Model for Real-World X-ray Baggage Security Inspection","date":"2025-04-03","arxiv_id":"2504.02823","repositories_listed":1,"syntology":null},{"url":"/paper/gmai-vl-r1-harnessing-reinforcement-learning","slug":"gmai-vl-r1-harnessing-reinforcement-learning","title":"GMAI-VL-R1: Harnessing Reinforcement Learning for Multimodal Medical Reasoning","date":"2025-04-02","arxiv_id":"2504.01886","repositories_listed":1,"syntology":null},{"url":"/paper/fortisavqa-and-maven-a-benchmark-dataset-and","slug":"fortisavqa-and-maven-a-benchmark-dataset-and","title":"FortisAVQA and MAVEN: a Benchmark Dataset and Debiasing Framework for Robust Multimodal Reasoning","date":"2025-04-01","arxiv_id":"2504.00487","repositories_listed":1,"syntology":null},{"url":"/paper/koffvqa-an-objectively-evaluated-free-form","slug":"koffvqa-an-objectively-evaluated-free-form","title":"KOFFVQA: An Objectively Evaluated Free-form VQA Benchmark for Large Vision-Language Models in the Korean Language","date":"2025-03-31","arxiv_id":"2503.23730","repositories_listed":1,"syntology":null},{"url":"/paper/opendrivevla-towards-end-to-end-autonomous","slug":"opendrivevla-towards-end-to-end-autonomous","title":"OpenDriveVLA: Towards End-to-end Autonomous Driving with Large Vision Language Action Model","date":"2025-03-30","arxiv_id":"2503.23463","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/opendrivevla-towards-end-to-end-autonomous#ran","syntology_url":"https://syntology.ai/paper/2503.23463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23463"}},"official":{"repos":["DriveVLA/OpenDriveVLA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/question-aware-knowledge-graph-prompting-for","slug":"question-aware-knowledge-graph-prompting-for","title":"Question-Aware Knowledge Graph Prompting for Enhancing Large Language Models","date":"2025-03-30","arxiv_id":"2503.23523","repositories_listed":1,"syntology":null},{"url":"/paper/refchartqa-grounding-visual-answer-on-chart","slug":"refchartqa-grounding-visual-answer-on-chart","title":"RefChartQA: Grounding Visual Answer on Chart Images through Instruction Tuning","date":"2025-03-29","arxiv_id":"2503.23131","repositories_listed":1,"syntology":null},{"url":"/paper/egotom-benchmarking-theory-of-mind-reasoning","slug":"egotom-benchmarking-theory-of-mind-reasoning","title":"EgoToM: Benchmarking Theory of Mind Reasoning from Egocentric Videos","date":"2025-03-28","arxiv_id":"2503.22152","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/egotom-benchmarking-theory-of-mind-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.22152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22152"}},"official":{"repos":["facebookresearch/egotom"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/preference-based-learning-with-retrieval","slug":"preference-based-learning-with-retrieval","title":"Preference-based Learning with Retrieval Augmented Generation for Conversational Question Answering","date":"2025-03-28","arxiv_id":"2503.22303","repositories_listed":1,"syntology":null},{"url":"/paper/facebench-a-multi-view-multi-level-facial","slug":"facebench-a-multi-view-multi-level-facial","title":"FaceBench: A Multi-View Multi-Level Facial Attribute VQA Dataset for Benchmarking Face Perception MLLMs","date":"2025-03-27","arxiv_id":"2503.21457","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/facebench-a-multi-view-multi-level-facial#ran","syntology_url":"https://syntology.ai/paper/2503.21457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21457"}},"official":{"repos":["cvi-szu/facebench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-evaluation-of-large-vision","slug":"fine-grained-evaluation-of-large-vision","title":"Fine-Grained Evaluation of Large Vision-Language Models in Autonomous Driving","date":"2025-03-27","arxiv_id":"2503.21505","repositories_listed":1,"syntology":null},{"url":"/paper/rearag-knowledge-guided-reasoning-enhances","slug":"rearag-knowledge-guided-reasoning-enhances","title":"ReaRAG: Knowledge-guided Reasoning Enhances Factuality of Large Reasoning Models with Iterative Retrieval Augmented Generation","date":"2025-03-27","arxiv_id":"2503.21729","repositories_listed":1,"syntology":null},{"url":"/paper/swi-speaking-with-intent-in-large-language","slug":"swi-speaking-with-intent-in-large-language","title":"SWI: Speaking with Intent in Large Language Models","date":"2025-03-27","arxiv_id":"2503.21544","repositories_listed":1,"syntology":null},{"url":"/paper/unified-multimodal-discrete-diffusion","slug":"unified-multimodal-discrete-diffusion","title":"Unified Multimodal Discrete Diffusion","date":"2025-03-26","arxiv_id":"2503.20853","repositories_listed":1,"syntology":null},{"url":"/paper/bibliopage-a-dataset-of-scanned-title-pages","slug":"bibliopage-a-dataset-of-scanned-title-pages","title":"BiblioPage: A Dataset of Scanned Title Pages for Bibliographic Metadata Extraction","date":"2025-03-25","arxiv_id":"2503.19658","repositories_listed":1,"syntology":null},{"url":"/paper/med3dvlm-an-efficient-vision-language-model","slug":"med3dvlm-an-efficient-vision-language-model","title":"Med3DVLM: An Efficient Vision-Language Model for 3D Medical Image Analysis","date":"2025-03-25","arxiv_id":"2503.20047","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/med3dvlm-an-efficient-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2503.20047","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20047"}},"official":{"repos":["mirthai/med3dvlm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/mind-the-gap-benchmarking-spatial-reasoning","slug":"mind-the-gap-benchmarking-spatial-reasoning","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","date":"2025-03-25","arxiv_id":"2503.19707","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-the-gap-benchmarking-spatial-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.19707","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19707"}},"official":{"repos":["stogiannidis/srbench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pave-patching-and-adapting-video-large","slug":"pave-patching-and-adapting-video-large","title":"PAVE: Patching and Adapting Video Large Language Models","date":"2025-03-25","arxiv_id":"2503.19794","repositories_listed":1,"syntology":null},{"url":"/paper/vgat-a-cancer-survival-analysis-framework","slug":"vgat-a-cancer-survival-analysis-framework","title":"VGAT: A Cancer Survival Analysis Framework Transitioning from Generative Visual Question Answering to Genomic Reconstruction","date":"2025-03-25","arxiv_id":"2503.19367","repositories_listed":1,"syntology":null},{"url":"/paper/llavaction-evaluating-and-training-multi","slug":"llavaction-evaluating-and-training-multi","title":"LLaVAction: evaluating and training multi-modal large language models for action recognition","date":"2025-03-24","arxiv_id":"2503.18712","repositories_listed":1,"syntology":null},{"url":"/paper/mc-llava-multi-concept-personalized-vision-1","slug":"mc-llava-multi-concept-personalized-vision-1","title":"MC-LLaVA: Multi-Concept Personalized Vision-Language Model","date":"2025-03-24","arxiv_id":"2503.18854","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-augmented-generation-and-1","slug":"retrieval-augmented-generation-and-1","title":"Retrieval Augmented Generation and Understanding in Vision: A Survey and New Outlook","date":"2025-03-23","arxiv_id":"2503.18016","repositories_listed":1,"syntology":null},{"url":"/paper/4d-bench-benchmarking-multi-modal-large","slug":"4d-bench-benchmarking-multi-modal-large","title":"4D-Bench: Benchmarking Multi-modal Large Language Models for 4D Object Understanding","date":"2025-03-22","arxiv_id":"2503.17827","repositories_listed":1,"syntology":null},{"url":"/paper/progressive-prompt-detailing-for-improved","slug":"progressive-prompt-detailing-for-improved","title":"Progressive Prompt Detailing for Improved Alignment in Text-to-Image Generative Models","date":"2025-03-22","arxiv_id":"2503.17794","repositories_listed":1,"syntology":null},{"url":"/paper/relation-extraction-with-instance-adapted","slug":"relation-extraction-with-instance-adapted","title":"Relation Extraction with Instance-Adapted Predicate Descriptions","date":"2025-03-22","arxiv_id":"2503.17799","repositories_listed":1,"syntology":null},{"url":"/paper/chain-of-tools-utilizing-massive-unseen-tools","slug":"chain-of-tools-utilizing-massive-unseen-tools","title":"Chain-of-Tools: Utilizing Massive Unseen Tools in the CoT Reasoning of Frozen Language Models","date":"2025-03-21","arxiv_id":"2503.16779","repositories_listed":1,"syntology":null},{"url":"/paper/dense-passage-retrieval-in-conversational","slug":"dense-passage-retrieval-in-conversational","title":"Dense Passage Retrieval in Conversational Search","date":"2025-03-21","arxiv_id":"2503.17507","repositories_listed":1,"syntology":null},{"url":"/paper/does-chain-of-thought-reasoning-help-mobile","slug":"does-chain-of-thought-reasoning-help-mobile","title":"Does Chain-of-Thought Reasoning Help Mobile GUI Agent? An Empirical Study","date":"2025-03-21","arxiv_id":"2503.16788","repositories_listed":1,"syntology":null},{"url":"/paper/mtbench-a-multimodal-time-series-benchmark","slug":"mtbench-a-multimodal-time-series-benchmark","title":"MTBench: A Multimodal Time Series Benchmark for Temporal Reasoning and Question Answering","date":"2025-03-21","arxiv_id":"2503.16858","repositories_listed":1,"syntology":null},{"url":"/paper/agentic-keyframe-search-for-video-question","slug":"agentic-keyframe-search-for-video-question","title":"Agentic Keyframe Search for Video Question Answering","date":"2025-03-20","arxiv_id":"2503.16032","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agentic-keyframe-search-for-video-question#ran","syntology_url":"https://syntology.ai/paper/2503.16032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16032"}},"official":{"repos":["fansunqi/akeys"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/typed-rag-type-aware-multi-aspect","slug":"typed-rag-type-aware-multi-aspect","title":"Typed-RAG: Type-aware Multi-Aspect Decomposition for Non-Factoid Question Answering","date":"2025-03-20","arxiv_id":"2503.15879","repositories_listed":1,"syntology":null},{"url":"/paper/umit-unifying-medical-imaging-tasks-via","slug":"umit-unifying-medical-imaging-tasks-via","title":"UMIT: Unifying Medical Imaging Tasks via Vision-Language Models","date":"2025-03-20","arxiv_id":"2503.15892","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-retrieval-strategies-for-financial","slug":"optimizing-retrieval-strategies-for-financial","title":"Optimizing Retrieval Strategies for Financial Question Answering Documents in Retrieval-Augmented Generation Systems","date":"2025-03-19","arxiv_id":"2503.15191","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/optimizing-retrieval-strategies-for-financial#ran","syntology_url":"https://syntology.ai/paper/2503.15191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.15191"}},"official":{"repos":["seohyunwoo-0407/gar"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/solla-towards-a-speech-oriented-llm-that","slug":"solla-towards-a-speech-oriented-llm-that","title":"Solla: Towards a Speech-Oriented LLM That Hears Acoustic Context","date":"2025-03-19","arxiv_id":"2503.15338","repositories_listed":1,"syntology":null},{"url":"/paper/how-much-do-llms-learn-from-negative-examples","slug":"how-much-do-llms-learn-from-negative-examples","title":"How much do LLMs learn from negative examples?","date":"2025-03-18","arxiv_id":"2503.14391","repositories_listed":1,"syntology":null},{"url":"/paper/mdocagent-a-multi-modal-multi-agent-framework","slug":"mdocagent-a-multi-modal-multi-agent-framework","title":"MDocAgent: A Multi-Modal Multi-Agent Framework for Document Understanding","date":"2025-03-18","arxiv_id":"2503.13964","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdocagent-a-multi-modal-multi-agent-framework#ran","syntology_url":"https://syntology.ai/paper/2503.13964","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13964"}},"official":{"repos":["aiming-lab/mdocagent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/where-do-large-vision-language-models-look-at","slug":"where-do-large-vision-language-models-look-at","title":"Where do Large Vision-Language Models Look at when Answering Questions?","date":"2025-03-18","arxiv_id":"2503.13891","repositories_listed":1,"syntology":null},{"url":"/paper/hicd-hallucination-inducing-via-attention","slug":"hicd-hallucination-inducing-via-attention","title":"HICD: Hallucination-Inducing via Attention Dispersion for Contrastive Decoding to Mitigate Hallucinations in Large Language Models","date":"2025-03-17","arxiv_id":"2503.12908","repositories_listed":1,"syntology":null},{"url":"/paper/mes-rag-bringing-multi-modal-entity-storage","slug":"mes-rag-bringing-multi-modal-entity-storage","title":"MES-RAG: Bringing Multi-modal, Entity-Storage, and Secure Enhancements to RAG","date":"2025-03-17","arxiv_id":"2503.13563","repositories_listed":1,"syntology":null},{"url":"/paper/microvqa-a-multimodal-reasoning-benchmark-for","slug":"microvqa-a-multimodal-reasoning-benchmark-for","title":"MicroVQA: A Multimodal Reasoning Benchmark for Microscopy-Based Scientific Research","date":"2025-03-17","arxiv_id":"2503.13399","repositories_listed":1,"syntology":null},{"url":"/paper/nuplanqa-a-large-scale-dataset-and-benchmark","slug":"nuplanqa-a-large-scale-dataset-and-benchmark","title":"NuPlanQA: A Large-Scale Dataset and Benchmark for Multi-View Driving Scene Understanding in Multi-Modal Large Language Models","date":"2025-03-17","arxiv_id":"2503.12772","repositories_listed":1,"syntology":null},{"url":"/paper/omnia-de-egotempo-benchmarking-temporal","slug":"omnia-de-egotempo-benchmarking-temporal","title":"Omnia de EgoTempo: Benchmarking Temporal Understanding of Multi-Modal LLMs in Egocentric Videos","date":"2025-03-17","arxiv_id":"2503.13646","repositories_listed":1,"syntology":null},{"url":"/paper/videomind-a-chain-of-lora-agent-for-long","slug":"videomind-a-chain-of-lora-agent-for-long","title":"VideoMind: A Chain-of-LoRA Agent for Long Video Reasoning","date":"2025-03-17","arxiv_id":"2503.13444","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videomind-a-chain-of-lora-agent-for-long#ran","syntology_url":"https://syntology.ai/paper/2503.13444","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13444"}},"official":{"repos":["yeliudev/VideoMind"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/t2i-fineeval-fine-grained-compositional","slug":"t2i-fineeval-fine-grained-compositional","title":"T2I-FineEval: Fine-Grained Compositional Metric for Text-to-Image Evaluation","date":"2025-03-14","arxiv_id":"2503.11481","repositories_listed":1,"syntology":null},{"url":"/paper/drivelmm-o1-a-step-by-step-reasoning-dataset","slug":"drivelmm-o1-a-step-by-step-reasoning-dataset","title":"DriveLMM-o1: A Step-by-Step Reasoning Dataset and Large Multimodal Model for Driving Scenario Understanding","date":"2025-03-13","arxiv_id":"2503.10621","repositories_listed":1,"syntology":null},{"url":"/paper/how-do-multimodal-large-language-models","slug":"how-do-multimodal-large-language-models","title":"How Do Multimodal Large Language Models Handle Complex Multimodal Reasoning? Placing Them in An Extensible Escape Game","date":"2025-03-13","arxiv_id":"2503.10042","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-augmented-generation-with-1","slug":"retrieval-augmented-generation-with-1","title":"Retrieval-Augmented Generation with Hierarchical Knowledge","date":"2025-03-13","arxiv_id":"2503.10150","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/retrieval-augmented-generation-with-1#ran","syntology_url":"https://syntology.ai/paper/2503.10150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10150"}},"official":{"repos":["hhy-huang/HiRAG"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/simlingo-vision-only-closed-loop-autonomous","slug":"simlingo-vision-only-closed-loop-autonomous","title":"SimLingo: Vision-Only Closed-Loop Autonomous Driving with Language-Action Alignment","date":"2025-03-12","arxiv_id":"2503.09594","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":8,"n_ran_checked":8,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 8 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/simlingo-vision-only-closed-loop-autonomous#ran","syntology_url":"https://syntology.ai/paper/2503.09594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.09594"}},"official":null}},{"url":"/paper/teaching-lmms-for-image-quality-scoring-and","slug":"teaching-lmms-for-image-quality-scoring-and","title":"Teaching LMMs for Image Quality Scoring and Interpreting","date":"2025-03-12","arxiv_id":"2503.09197","repositories_listed":1,"syntology":null},{"url":"/paper/plainqafact-automatic-factuality-evaluation","slug":"plainqafact-automatic-factuality-evaluation","title":"PlainQAFact: Automatic Factuality Evaluation Metric for Biomedical Plain Language Summaries Generation","date":"2025-03-11","arxiv_id":"2503.08890","repositories_listed":1,"syntology":null},{"url":"/paper/a-multimodal-benchmark-dataset-and-model-for","slug":"a-multimodal-benchmark-dataset-and-model-for","title":"A Multimodal Benchmark Dataset and Model for Crop Disease Diagnosis","date":"2025-03-10","arxiv_id":"2503.06973","repositories_listed":1,"syntology":null},{"url":"/paper/kwaichat-a-large-scale-video-driven","slug":"kwaichat-a-large-scale-video-driven","title":"KwaiChat: A Large-Scale Video-Driven Multilingual Mixed-Type Dialogue Corpus","date":"2025-03-10","arxiv_id":"2503.06899","repositories_listed":1,"syntology":null},{"url":"/paper/medagentsbench-benchmarking-thinking-models","slug":"medagentsbench-benchmarking-thinking-models","title":"MedAgentsBench: Benchmarking Thinking Models and Agent Frameworks for Complex Medical Reasoning","date":"2025-03-10","arxiv_id":"2503.07459","repositories_listed":1,"syntology":{"n":23,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":4,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/medagentsbench-benchmarking-thinking-models#ran","syntology_url":"https://syntology.ai/paper/2503.07459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07459"}},"official":{"repos":["gersteinlab/medagents-benchmark"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/treble-counterfactual-vlms-a-causal-approach","slug":"treble-counterfactual-vlms-a-causal-approach","title":"Treble Counterfactual VLMs: A Causal Approach to Hallucination","date":"2025-03-08","arxiv_id":"2503.06169","repositories_listed":1,"syntology":null},{"url":"/paper/anyanomaly-zero-shot-customizable-video-1","slug":"anyanomaly-zero-shot-customizable-video-1","title":"AnyAnomaly: Zero-Shot Customizable Video Anomaly Detection with LVLM","date":"2025-03-06","arxiv_id":"2503.04504","repositories_listed":1,"syntology":null},{"url":"/paper/keeping-yourself-is-important-in-downstream","slug":"keeping-yourself-is-important-in-downstream","title":"Keeping Yourself is Important in Downstream Tuning Multimodal Large Language Model","date":"2025-03-06","arxiv_id":"2503.04543","repositories_listed":1,"syntology":null},{"url":"/paper/question-aware-gaussian-experts-for-audio","slug":"question-aware-gaussian-experts-for-audio","title":"Question-Aware Gaussian Experts for Audio-Visual Question Answering","date":"2025-03-06","arxiv_id":"2503.04459","repositories_listed":1,"syntology":null},{"url":"/paper/robust-data-watermarking-in-language-models","slug":"robust-data-watermarking-in-language-models","title":"Robust Data Watermarking in Language Models by Injecting Fictitious Knowledge","date":"2025-03-06","arxiv_id":"2503.04036","repositories_listed":1,"syntology":null},{"url":"/paper/addressing-overprescribing-challenges-fine","slug":"addressing-overprescribing-challenges-fine","title":"Addressing Overprescribing Challenges: Fine-Tuning Large Language Models for Medication Recommendation Tasks","date":"2025-03-05","arxiv_id":"2503.03687","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/addressing-overprescribing-challenges-fine#ran","syntology_url":"https://syntology.ai/paper/2503.03687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03687"}},"official":{"repos":["zzhustc2016/lamo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/attackseqbench-benchmarking-large-language","slug":"attackseqbench-benchmarking-large-language","title":"AttackSeqBench: Benchmarking Large Language Models' Understanding of Sequential Patterns in Cyber Attacks","date":"2025-03-05","arxiv_id":"2503.03170","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-causal-relation-alignment-for-1","slug":"cross-modal-causal-relation-alignment-for-1","title":"Cross-modal Causal Relation Alignment for Video Question Grounding","date":"2025-03-05","arxiv_id":"2503.07635","repositories_listed":1,"syntology":{"n":16,"n_ran":13,"n_constructed":12,"n_ran_checked":13,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":16,"phrase":"13 ran (of which 12 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cross-modal-causal-relation-alignment-for-1#ran","syntology_url":"https://syntology.ai/paper/2503.07635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07635"}},"official":{"repos":["wissingchen/cra-gqa"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":12,"n_ran_no_instrument_failure":13,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/dspnet-dual-vision-scene-perception-for","slug":"dspnet-dual-vision-scene-perception-for","title":"DSPNet: Dual-vision Scene Perception for Robust 3D Question Answering","date":"2025-03-05","arxiv_id":"2503.03190","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dspnet-dual-vision-scene-perception-for#ran","syntology_url":"https://syntology.ai/paper/2503.03190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03190"}},"official":{"repos":["LZ-CH/DSPNet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/egolife-towards-egocentric-life-assistant","slug":"egolife-towards-egocentric-life-assistant","title":"EgoLife: Towards Egocentric Life Assistant","date":"2025-03-05","arxiv_id":"2503.03803","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":9,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/egolife-towards-egocentric-life-assistant#ran","syntology_url":"https://syntology.ai/paper/2503.03803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03803"}},"official":{"repos":["evolvinglmms-lab/egolife"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-vietnamese-vqa-through-curriculum","slug":"enhancing-vietnamese-vqa-through-curriculum","title":"Enhancing Vietnamese VQA through Curriculum Learning on Raw and Augmented Text Representations","date":"2025-03-05","arxiv_id":"2503.03285","repositories_listed":1,"syntology":null},{"url":"/paper/task-agnostic-attacks-against-vision","slug":"task-agnostic-attacks-against-vision","title":"Task-Agnostic Attacks Against Vision Foundation Models","date":"2025-03-05","arxiv_id":"2503.03842","repositories_listed":1,"syntology":null},{"url":"/paper/biod2c-a-dual-level-semantic-consistency","slug":"biod2c-a-dual-level-semantic-consistency","title":"BioD2C: A Dual-level Semantic Consistency Constraint Framework for Biomedical VQA","date":"2025-03-04","arxiv_id":"2503.02476","repositories_listed":1,"syntology":null},{"url":"/paper/towards-robust-expert-finding-in-community","slug":"towards-robust-expert-finding-in-community","title":"Towards Robust Expert Finding in Community Question Answering Platforms","date":"2025-03-04","arxiv_id":"2503.02674","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-complex-question-answering-on-long","slug":"zero-shot-complex-question-answering-on-long","title":"Zero-Shot Complex Question-Answering on Long Scientific Documents","date":"2025-03-04","arxiv_id":"2503.02695","repositories_listed":1,"syntology":null},{"url":"/paper/q-nl-verifier-leveraging-synthetic-data-for","slug":"q-nl-verifier-leveraging-synthetic-data-for","title":"Q-NL Verifier: Leveraging Synthetic Data for Robust Knowledge Graph Question Answering","date":"2025-03-03","arxiv_id":"2503.01385","repositories_listed":1,"syntology":null},{"url":"/paper/watch-out-your-album-on-the-inadvertent","slug":"watch-out-your-album-on-the-inadvertent","title":"Watch Out Your Album! On the Inadvertent Privacy Memorization in Multi-Modal Large Language Models","date":"2025-03-03","arxiv_id":"2503.01208","repositories_listed":1,"syntology":null},{"url":"/paper/when-an-llm-is-apprehensive-about-its-answers","slug":"when-an-llm-is-apprehensive-about-its-answers","title":"When an LLM is apprehensive about its answers -- and when its uncertainty is justified","date":"2025-03-03","arxiv_id":"2503.01688","repositories_listed":1,"syntology":null},{"url":"/paper/2503-00955","slug":"2503-00955","title":"SemViQA: A Semantic Question Answering System for Vietnamese Information Fact-Checking","date":"2025-03-02","arxiv_id":"2503.00955","repositories_listed":1,"syntology":null},{"url":"/paper/ails-ntua-at-semeval-2025-task-8-language-to","slug":"ails-ntua-at-semeval-2025-task-8-language-to","title":"AILS-NTUA at SemEval-2025 Task 8: Language-to-Code prompting and Error Fixing for Tabular Question Answering","date":"2025-03-01","arxiv_id":"2503.00435","repositories_listed":1,"syntology":null},{"url":"/paper/glossgpt-gpt-for-word-sense-disambiguation","slug":"glossgpt-gpt-for-word-sense-disambiguation","title":"GlossGPT: GPT for Word Sense Disambiguation using Few-shot Chain-of-Thought Prompting","date":"2025-03-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/streaming-video-question-answering-with-in","slug":"streaming-video-question-answering-with-in","title":"Streaming Video Question-Answering with In-context Video KV-Cache Retrieval","date":"2025-03-01","arxiv_id":"2503.00540","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":6,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/streaming-video-question-answering-with-in#ran","syntology_url":"https://syntology.ai/paper/2503.00540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00540"}},"official":{"repos":["becomebright/rekv"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/medhalltune-an-instruction-tuning-benchmark","slug":"medhalltune-an-instruction-tuning-benchmark","title":"MedHallTune: An Instruction-Tuning Benchmark for Mitigating Medical Hallucination in Vision-Language Models","date":"2025-02-28","arxiv_id":"2502.20780","repositories_listed":1,"syntology":null},{"url":"/paper/chineseecomqa-a-scalable-e-commerce-concept","slug":"chineseecomqa-a-scalable-e-commerce-concept","title":"ChineseEcomQA: A Scalable E-commerce Concept Evaluation Benchmark for Large Language Models","date":"2025-02-27","arxiv_id":"2502.20196","repositories_listed":1,"syntology":null},{"url":"/paper/few-shot-multilingual-open-domain-qa-from-5","slug":"few-shot-multilingual-open-domain-qa-from-5","title":"Few-Shot Multilingual Open-Domain QA from 5 Examples","date":"2025-02-27","arxiv_id":"2502.19722","repositories_listed":1,"syntology":null},{"url":"/paper/protecting-multimodal-large-language-models","slug":"protecting-multimodal-large-language-models","title":"Protecting multimodal large language models against misleading visualizations","date":"2025-02-27","arxiv_id":"2502.20503","repositories_listed":1,"syntology":null},{"url":"/paper/fspo-few-shot-preference-optimization-of","slug":"fspo-few-shot-preference-optimization-of","title":"FSPO: Few-Shot Preference Optimization of Synthetic Preference Data in LLMs Elicits Effective Personalization to Real Users","date":"2025-02-26","arxiv_id":"2502.19312","repositories_listed":1,"syntology":null},{"url":"/paper/uqabench-evaluating-user-embedding-for","slug":"uqabench-evaluating-user-embedding-for","title":"UQABench: Evaluating User Embedding for Prompting LLMs in Personalized Question Answering","date":"2025-02-26","arxiv_id":"2502.19178","repositories_listed":1,"syntology":null}],"record_sha256":"476bbd35252371aefc8c966d2a2d03d94472946eda1a252c5fe393779867604a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}