{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/3","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":22,"rows_per_page":100,"rows":[201,300],"of":2167,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering","prev":"/task/visual-question-answering/papers/2","next":"/task/visual-question-answering/papers/4","papers":[{"url":"/paper/counterfactual-samples-synthesizing-for","slug":"counterfactual-samples-synthesizing-for","title":"Counterfactual Samples Synthesizing for Robust Visual Question Answering","date":"2020-03-14","arxiv_id":"2003.06576","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-samples-synthesizing-for#ran","syntology_url":"https://syntology.ai/paper/2003.06576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06576"}},"official":{"repos":["yanxinzju/CSS-VQA"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/fine-grained-image-classification-and","slug":"fine-grained-image-classification-and","title":"Fine-grained Image Classification and Retrieval by Combining Visual and Locally Pooled Textual Features","date":"2020-01-14","arxiv_id":"2001.04732","repositories_listed":2,"syntology":null},{"url":"/paper/in-defense-of-grid-features-for-visual","slug":"in-defense-of-grid-features-for-visual","title":"In Defense of Grid Features for Visual Question Answering","date":"2020-01-10","arxiv_id":"2001.03615","repositories_listed":2,"syntology":null},{"url":"/paper/overcoming-data-limitation-in-medical-visual","slug":"overcoming-data-limitation-in-medical-visual","title":"Overcoming Data Limitation in Medical Visual Question Answering","date":"2019-09-26","arxiv_id":"1909.11867","repositories_listed":2,"syntology":null},{"url":"/paper/omninet-a-unified-architecture-for-multi","slug":"omninet-a-unified-architecture-for-multi","title":"OmniNet: A unified architecture for multi-modal multi-task learning","date":"2019-07-17","arxiv_id":"1907.07804","repositories_listed":2,"syntology":null},{"url":"/paper/the-neuro-symbolic-concept-learner-1","slug":"the-neuro-symbolic-concept-learner-1","title":"The Neuro-Symbolic Concept Learner: Interpreting Scenes, Words, and Sentences From Natural Supervision","date":"2019-04-26","arxiv_id":"1904.12584","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-neuro-symbolic-concept-learner-1#ran","syntology_url":"https://syntology.ai/paper/1904.12584","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.12584"}},"official":{"repos":["vacancy/NSCL-PyTorch-Release"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/answer-them-all-toward-universal-visual","slug":"answer-them-all-toward-universal-visual","title":"Answer Them All! Toward Universal Visual Question Answering Models","date":"2019-03-01","arxiv_id":"1903.00366","repositories_listed":2,"syntology":null},{"url":"/paper/dual-attention-networks-for-visual-reference","slug":"dual-attention-networks-for-visual-reference","title":"Dual Attention Networks for Visual Reference Resolution in Visual Dialog","date":"2019-02-25","arxiv_id":"1902.09368","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dual-attention-networks-for-visual-reference#ran","syntology_url":"https://syntology.ai/paper/1902.09368","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1902.09368"}},"official":{"repos":["gicheonkang/DAN-VisDial"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learning-representations-of-sets-through","slug":"learning-representations-of-sets-through","title":"Learning Representations of Sets through Optimized Permutations","date":"2018-12-10","arxiv_id":"1812.03928","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-representations-of-sets-through#ran","syntology_url":"https://syntology.ai/paper/1812.03928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.03928"}},"official":{"repos":["Cyanogenoid/perm-optim","iclr2019-anon123456/perm-optim"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/explainable-and-explicit-visual-reasoning","slug":"explainable-and-explicit-visual-reasoning","title":"Explainable and Explicit Visual Reasoning over Scene Graphs","date":"2018-12-05","arxiv_id":"1812.01855","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/explainable-and-explicit-visual-reasoning#ran","syntology_url":"https://syntology.ai/paper/1812.01855","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1812.01855"}},"official":{"repos":["shijx12/XNM-Net"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/systematic-generalization-what-is-required","slug":"systematic-generalization-what-is-required","title":"Systematic Generalization: What Is Required and Can It Be Learned?","date":"2018-11-30","arxiv_id":"1811.12889","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/systematic-generalization-what-is-required#ran","syntology_url":"https://syntology.ai/paper/1811.12889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.12889"}},"official":{"repos":["rizar/systematic-generalization-sqoop"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","unlocated"]}}},{"url":"/paper/neural-symbolic-vqa-disentangling-reasoning","slug":"neural-symbolic-vqa-disentangling-reasoning","title":"Neural-Symbolic VQA: Disentangling Reasoning from Vision and Language Understanding","date":"2018-10-04","arxiv_id":"1810.02338","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neural-symbolic-vqa-disentangling-reasoning#ran","syntology_url":"https://syntology.ai/paper/1810.02338","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02338"}},"official":null}},{"url":"/paper/a-joint-sequence-fusion-model-for-video","slug":"a-joint-sequence-fusion-model-for-video","title":"A Joint Sequence Fusion Model for Video Question Answering and Retrieval","date":"2018-08-07","arxiv_id":"1808.02559","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-joint-sequence-fusion-model-for-video#ran","syntology_url":"https://syntology.ai/paper/1808.02559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1808.02559"}},"official":null}},{"url":"/paper/end-to-end-audio-visual-scene-aware-dialog","slug":"end-to-end-audio-visual-scene-aware-dialog","title":"End-to-End Audio Visual Scene-Aware Dialog using Multimodal Attention-Based Video Features","date":"2018-06-21","arxiv_id":"1806.08409","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/end-to-end-audio-visual-scene-aware-dialog#ran","syntology_url":"https://syntology.ai/paper/1806.08409","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.08409"}},"official":null}},{"url":"/paper/focal-visual-text-attention-for-visual","slug":"focal-visual-text-attention-for-visual","title":"Focal Visual-Text Attention for Visual Question Answering","date":"2018-06-05","arxiv_id":"1806.01873","repositories_listed":2,"syntology":null},{"url":"/paper/ai2-thor-an-interactive-3d-environment-for","slug":"ai2-thor-an-interactive-3d-environment-for","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","date":"2017-12-14","arxiv_id":"1712.05474","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ai2-thor-an-interactive-3d-environment-for#ran","syntology_url":"https://syntology.ai/paper/1712.05474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1712.05474"}},"official":{"repos":["allenai/ai2thor"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-bilinear-generalized-multi-modal","slug":"beyond-bilinear-generalized-multi-modal","title":"Beyond Bilinear: Generalized Multimodal Factorized High-order Pooling for Visual Question Answering","date":"2017-08-10","arxiv_id":"1708.03619","repositories_listed":2,"syntology":null},{"url":"/paper/modulating-early-visual-processing-by","slug":"modulating-early-visual-processing-by","title":"Modulating early visual processing by language","date":"2017-07-02","arxiv_id":"1707.00683","repositories_listed":2,"syntology":null},{"url":"/paper/tgif-qa-toward-spatio-temporal-reasoning-in","slug":"tgif-qa-toward-spatio-temporal-reasoning-in","title":"TGIF-QA: Toward Spatio-Temporal Reasoning in Visual Question Answering","date":"2017-04-14","arxiv_id":"1704.04497","repositories_listed":2,"syntology":null},{"url":"/paper/end-to-end-optimization-of-goal-driven-and","slug":"end-to-end-optimization-of-goal-driven-and","title":"End-to-end optimization of goal-driven and visually grounded dialogue systems","date":"2017-03-15","arxiv_id":"1703.05423","repositories_listed":2,"syntology":null},{"url":"/paper/modeling-relationships-in-referential","slug":"modeling-relationships-in-referential","title":"Modeling Relationships in Referential Expressions with Compositional Modular Networks","date":"2016-11-30","arxiv_id":"1611.09978","repositories_listed":2,"syntology":null},{"url":"/paper/grad-cam-why-did-you-say-that","slug":"grad-cam-why-did-you-say-that","title":"Grad-CAM: Why did you say that?","date":"2016-11-22","arxiv_id":"1611.07450","repositories_listed":2,"syntology":null},{"url":"/paper/dual-attention-networks-for-multimodal","slug":"dual-attention-networks-for-multimodal","title":"Dual Attention Networks for Multimodal Reasoning and Matching","date":"2016-11-02","arxiv_id":"1611.00471","repositories_listed":2,"syntology":null},{"url":"/paper/neural-module-networks","slug":"neural-module-networks","title":"Neural Module Networks","date":"2015-11-09","arxiv_id":"1511.02799","repositories_listed":2,"syntology":null},{"url":"/paper/describe-anything-model-for-visual-question","slug":"describe-anything-model-for-visual-question","title":"Describe Anything Model for Visual Question Answering on Text-rich Images","date":"2025-07-16","arxiv_id":"2507.12441","repositories_listed":1,"syntology":null},{"url":"/paper/decoupled-seg-tokens-make-stronger-reasoning","slug":"decoupled-seg-tokens-make-stronger-reasoning","title":"Decoupled Seg Tokens Make Stronger Reasoning Video Segmenter and Grounder","date":"2025-06-28","arxiv_id":"2506.22880","repositories_listed":1,"syntology":null},{"url":"/paper/drishtikon-multi-granular-visual-grounding","slug":"drishtikon-multi-granular-visual-grounding","title":"DrishtiKon: Multi-Granular Visual Grounding for Text-Rich Document Images","date":"2025-06-26","arxiv_id":"2506.21316","repositories_listed":1,"syntology":null},{"url":"/paper/hribench-benchmarking-vision-language-models","slug":"hribench-benchmarking-vision-language-models","title":"HRIBench: Benchmarking Vision-Language Models for Real-Time Human Perception in Human-Robot Interaction","date":"2025-06-25","arxiv_id":"2506.20566","repositories_listed":1,"syntology":null},{"url":"/paper/mmsearch-r1-incentivizing-lmms-to-search","slug":"mmsearch-r1-incentivizing-lmms-to-search","title":"MMSearch-R1: Incentivizing LMMs to Search","date":"2025-06-25","arxiv_id":"2506.20670","repositories_listed":1,"syntology":null},{"url":"/paper/adapting-lightweight-vision-language-models","slug":"adapting-lightweight-vision-language-models","title":"Adapting Lightweight Vision Language Models for Radiological Visual Question Answering","date":"2025-06-17","arxiv_id":"2506.14451","repositories_listed":1,"syntology":null},{"url":"/paper/slotpi-physics-informed-object-centric","slug":"slotpi-physics-informed-object-centric","title":"SlotPi: Physics-informed Object-centric Reasoning Models","date":"2025-06-12","arxiv_id":"2506.10778","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/slotpi-physics-informed-object-centric#ran","syntology_url":"https://syntology.ai/paper/2506.10778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.10778"}},"official":{"repos":["intell-sci-comput/slotpi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/causalvqa-a-physically-grounded-causal","slug":"causalvqa-a-physically-grounded-causal","title":"CausalVQA: A Physically Grounded Causal Reasoning Benchmark for Video Models","date":"2025-06-11","arxiv_id":"2506.09943","repositories_listed":1,"syntology":null},{"url":"/paper/outside-knowledge-conversational-video-okcv","slug":"outside-knowledge-conversational-video-okcv","title":"Outside Knowledge Conversational Video (OKCV) Dataset -- Dialoguing over Videos","date":"2025-06-11","arxiv_id":"2506.09953","repositories_listed":1,"syntology":null},{"url":"/paper/haibu-remud-reasoning-multimodal-ultrasound","slug":"haibu-remud-reasoning-multimodal-ultrasound","title":"HAIBU-ReMUD: Reasoning Multimodal Ultrasound Dataset and Model Bridging to General Specific Domains","date":"2025-06-09","arxiv_id":"2506.07837","repositories_listed":1,"syntology":null},{"url":"/paper/looking-beyond-visible-cues-implicit-video","slug":"looking-beyond-visible-cues-implicit-video","title":"Looking Beyond Visible Cues: Implicit Video Question Answering via Dual-Clue Reasoning","date":"2025-06-09","arxiv_id":"2506.07811","repositories_listed":1,"syntology":null},{"url":"/paper/videocad-a-large-scale-video-dataset-for","slug":"videocad-a-large-scale-video-dataset-for","title":"VideoCAD: A Large-Scale Video Dataset for Learning UI Interactions and 3D Reasoning from CAD Software","date":"2025-05-30","arxiv_id":"2505.24838","repositories_listed":1,"syntology":null},{"url":"/paper/interpreting-chest-x-rays-like-a-radiologist","slug":"interpreting-chest-x-rays-like-a-radiologist","title":"Interpreting Chest X-rays Like a Radiologist: A Benchmark with Clinical Reasoning","date":"2025-05-29","arxiv_id":"2505.23143","repositories_listed":1,"syntology":null},{"url":"/paper/multi-sourced-compositional-generalization-in","slug":"multi-sourced-compositional-generalization-in","title":"Multi-Sourced Compositional Generalization in Visual Question Answering","date":"2025-05-29","arxiv_id":"2505.23045","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-document-question-answering-in","slug":"synthetic-document-question-answering-in","title":"Synthetic Document Question Answering in Hungarian","date":"2025-05-29","arxiv_id":"2505.23008","repositories_listed":1,"syntology":null},{"url":"/paper/vignette-socially-grounded-bias-evaluation","slug":"vignette-socially-grounded-bias-evaluation","title":"VIGNETTE: Socially Grounded Bias Evaluation for Vision-Language Models","date":"2025-05-28","arxiv_id":"2505.22897","repositories_listed":1,"syntology":null},{"url":"/paper/frames-vqa-benchmarking-fine-tuning-1","slug":"frames-vqa-benchmarking-fine-tuning-1","title":"FRAMES-VQA: Benchmarking Fine-Tuning Robustness across Multi-Modal Shifts in Visual Question Answering","date":"2025-05-27","arxiv_id":"2505.21755","repositories_listed":1,"syntology":null},{"url":"/paper/geollava-8k-scaling-remote-sensing-multimodal","slug":"geollava-8k-scaling-remote-sensing-multimodal","title":"GeoLLaVA-8K: Scaling Remote-Sensing Multimodal Large Language Models to 8K Resolution","date":"2025-05-27","arxiv_id":"2505.21375","repositories_listed":1,"syntology":null},{"url":"/paper/diagnosing-and-mitigating-modality","slug":"diagnosing-and-mitigating-modality","title":"Diagnosing and Mitigating Modality Interference in Multimodal Large Language Models","date":"2025-05-26","arxiv_id":"2505.19616","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diagnosing-and-mitigating-modality#ran","syntology_url":"https://syntology.ai/paper/2505.19616","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19616"}},"official":null}},{"url":"/paper/mineanybuild-benchmarking-spatial-planning","slug":"mineanybuild-benchmarking-spatial-planning","title":"MineAnyBuild: Benchmarking Spatial Planning for Open-world AI Agents","date":"2025-05-26","arxiv_id":"2505.20148","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mineanybuild-benchmarking-spatial-planning#ran","syntology_url":"https://syntology.ai/paper/2505.20148","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.20148"}},"official":{"repos":["mineanybuild/mineanybuild"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/unifying-multimodal-large-language-model","slug":"unifying-multimodal-large-language-model","title":"Unifying Multimodal Large Language Model Capabilities and Modalities via Model Merging","date":"2025-05-26","arxiv_id":"2505.19892","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2505.19892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19892"}},"official":{"repos":["walkerworldpeace/mllmerging"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/are-vision-language-models-ready-for-clinical","slug":"are-vision-language-models-ready-for-clinical","title":"Are Vision Language Models Ready for Clinical Diagnosis? A 3D Medical Benchmark for Tumor-centric Visual Question Answering","date":"2025-05-25","arxiv_id":"2505.18915","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-vision-language-models-ready-for-clinical#ran","syntology_url":"https://syntology.ai/paper/2505.18915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18915"}},"official":{"repos":["schuture/deeptumorvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/medical-large-vision-language-models-with","slug":"medical-large-vision-language-models-with","title":"Medical Large Vision Language Models with Multi-Image Visual Ability","date":"2025-05-25","arxiv_id":"2505.19031","repositories_listed":1,"syntology":null},{"url":"/paper/ntire-2025-challenge-on-video-quality","slug":"ntire-2025-challenge-on-video-quality","title":"NTIRE 2025 Challenge on Video Quality Enhancement for Video Conferencing: Datasets, Methods and Results","date":"2025-05-25","arxiv_id":"2505.18988","repositories_listed":1,"syntology":null},{"url":"/paper/satori-r1-incentivizing-multimodal-reasoning","slug":"satori-r1-incentivizing-multimodal-reasoning","title":"SATORI-R1: Incentivizing Multimodal Reasoning with Spatial Grounding and Verifiable Rewards","date":"2025-05-25","arxiv_id":"2505.19094","repositories_listed":1,"syntology":null},{"url":"/paper/let-androids-dream-of-electric-sheep-a-human","slug":"let-androids-dream-of-electric-sheep-a-human","title":"Let Androids Dream of Electric Sheep: A Human-like Image Implication Understanding and Reasoning Framework","date":"2025-05-22","arxiv_id":"2505.17019","repositories_listed":1,"syntology":null},{"url":"/paper/snap-a-benchmark-for-testing-the-effects-of","slug":"snap-a-benchmark-for-testing-the-effects-of","title":"SNAP: A Benchmark for Testing the Effects of Capture Conditions on Fundamental Vision Tasks","date":"2025-05-21","arxiv_id":"2505.15628","repositories_listed":1,"syntology":null},{"url":"/paper/medagentboard-benchmarking-multi-agent","slug":"medagentboard-benchmarking-multi-agent","title":"MedAgentBoard: Benchmarking Multi-Agent Collaboration with Conventional Methods for Diverse Medical Tasks","date":"2025-05-18","arxiv_id":"2505.12371","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/medagentboard-benchmarking-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2505.12371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12371"}},"official":{"repos":["yhzhu99/medagentboard"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/rvtbench-a-benchmark-for-visual-reasoning","slug":"rvtbench-a-benchmark-for-visual-reasoning","title":"RVTBench: A Benchmark for Visual Reasoning Tasks","date":"2025-05-17","arxiv_id":"2505.11838","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11454","slug":"2505-11454","title":"HumaniBench: A Human-Centric Framework for Large Multimodal Models Evaluation","date":"2025-05-16","arxiv_id":"2505.11454","repositories_listed":1,"syntology":null},{"url":"/paper/tcc-bench-benchmarking-the-traditional","slug":"tcc-bench-benchmarking-the-traditional","title":"TCC-Bench: Benchmarking the Traditional Chinese Culture Understanding Capabilities of MLLMs","date":"2025-05-16","arxiv_id":"2505.11275","repositories_listed":1,"syntology":null},{"url":"/paper/mm-skin-enhancing-dermatology-vision-language","slug":"mm-skin-enhancing-dermatology-vision-language","title":"MM-Skin: Enhancing Dermatology Vision-Language Model with an Image-Text Dataset Derived from Textbooks","date":"2025-05-09","arxiv_id":"2505.06152","repositories_listed":1,"syntology":null},{"url":"/paper/breaking-annotation-barriers-generalized","slug":"breaking-annotation-barriers-generalized","title":"Breaking Annotation Barriers: Generalized Video Quality Assessment via Ranking-based Self-Supervision","date":"2025-05-06","arxiv_id":"2505.03631","repositories_listed":1,"syntology":null},{"url":"/paper/adcare-vlm-leveraging-large-vision-language","slug":"adcare-vlm-leveraging-large-vision-language","title":"AdCare-VLM: Leveraging Large Vision Language Model (LVLM) to Monitor Long-Term Medication Adherence and Care","date":"2025-05-01","arxiv_id":"2505.00275","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/adcare-vlm-leveraging-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2505.00275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00275"}},"official":{"repos":["asad14053/AdCare-VLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/unlearning-sensitive-information-in","slug":"unlearning-sensitive-information-in","title":"Unlearning Sensitive Information in Multimodal LLMs: Benchmark and Attack-Defense Evaluation","date":"2025-05-01","arxiv_id":"2505.01456","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unlearning-sensitive-information-in#ran","syntology_url":"https://syntology.ai/paper/2505.01456","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.01456"}},"official":{"repos":["vaidehi99/unlok-vqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/videomultiagents-a-multi-agent-framework-for","slug":"videomultiagents-a-multi-agent-framework-for","title":"VideoMultiAgents: A Multi-Agent Framework for Video Question Answering","date":"2025-04-25","arxiv_id":"2504.20091","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-on-prompt-compression-for","slug":"an-empirical-study-on-prompt-compression-for","title":"An Empirical Study on Prompt Compression for Large Language Models","date":"2025-04-24","arxiv_id":"2505.00019","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-on-prompt-compression-for#ran","syntology_url":"https://syntology.ai/paper/2505.00019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.00019"}},"official":{"repos":["3DAgentWorld/Toolkit-for-Prompt-Compression"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ntire-2025-challenge-on-short-form-ugc-video","slug":"ntire-2025-challenge-on-short-form-ugc-video","title":"NTIRE 2025 Challenge on Short-form UGC Video Quality Assessment and Enhancement: Methods and Results","date":"2025-04-17","arxiv_id":"2504.13131","repositories_listed":1,"syntology":null},{"url":"/paper/qava-query-agnostic-visual-attack-to-large","slug":"qava-query-agnostic-visual-attack-to-large","title":"QAVA: Query-Agnostic Visual Attack to Large Vision-Language Models","date":"2025-04-15","arxiv_id":"2504.11038","repositories_listed":1,"syntology":null},{"url":"/paper/fvq-a-large-scale-dataset-and-a-lmm-based","slug":"fvq-a-large-scale-dataset-and-a-lmm-based","title":"FVQ: A Large-Scale Dataset and A LMM-based Method for Face Video Quality Assessment","date":"2025-04-12","arxiv_id":"2504.09255","repositories_listed":1,"syntology":null},{"url":"/paper/mimic-in-context-learning-for-multimodal","slug":"mimic-in-context-learning-for-multimodal","title":"Mimic In-Context Learning for Multimodal Tasks","date":"2025-04-11","arxiv_id":"2504.08851","repositories_listed":1,"syntology":null},{"url":"/paper/sting-bee-towards-vision-language-model-for","slug":"sting-bee-towards-vision-language-model-for","title":"STING-BEE: Towards Vision-Language Model for Real-World X-ray Baggage Security Inspection","date":"2025-04-03","arxiv_id":"2504.02823","repositories_listed":1,"syntology":null},{"url":"/paper/koffvqa-an-objectively-evaluated-free-form","slug":"koffvqa-an-objectively-evaluated-free-form","title":"KOFFVQA: An Objectively Evaluated Free-form VQA Benchmark for Large Vision-Language Models in the Korean Language","date":"2025-03-31","arxiv_id":"2503.23730","repositories_listed":1,"syntology":null},{"url":"/paper/facebench-a-multi-view-multi-level-facial","slug":"facebench-a-multi-view-multi-level-facial","title":"FaceBench: A Multi-View Multi-Level Facial Attribute VQA Dataset for Benchmarking Face Perception MLLMs","date":"2025-03-27","arxiv_id":"2503.21457","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/facebench-a-multi-view-multi-level-facial#ran","syntology_url":"https://syntology.ai/paper/2503.21457","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.21457"}},"official":{"repos":["cvi-szu/facebench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/med3dvlm-an-efficient-vision-language-model","slug":"med3dvlm-an-efficient-vision-language-model","title":"Med3DVLM: An Efficient Vision-Language Model for 3D Medical Image Analysis","date":"2025-03-25","arxiv_id":"2503.20047","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/med3dvlm-an-efficient-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2503.20047","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20047"}},"official":{"repos":["mirthai/med3dvlm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/vgat-a-cancer-survival-analysis-framework","slug":"vgat-a-cancer-survival-analysis-framework","title":"VGAT: A Cancer Survival Analysis Framework Transitioning from Generative Visual Question Answering to Genomic Reconstruction","date":"2025-03-25","arxiv_id":"2503.19367","repositories_listed":1,"syntology":null},{"url":"/paper/amd-hummingbird-towards-an-efficient-text-to","slug":"amd-hummingbird-towards-an-efficient-text-to","title":"AMD-Hummingbird: Towards an Efficient Text-to-Video Model","date":"2025-03-24","arxiv_id":"2503.18559","repositories_listed":1,"syntology":null},{"url":"/paper/progressive-prompt-detailing-for-improved","slug":"progressive-prompt-detailing-for-improved","title":"Progressive Prompt Detailing for Improved Alignment in Text-to-Image Generative Models","date":"2025-03-22","arxiv_id":"2503.17794","repositories_listed":1,"syntology":null},{"url":"/paper/microvqa-a-multimodal-reasoning-benchmark-for","slug":"microvqa-a-multimodal-reasoning-benchmark-for","title":"MicroVQA: A Multimodal Reasoning Benchmark for Microscopy-Based Scientific Research","date":"2025-03-17","arxiv_id":"2503.13399","repositories_listed":1,"syntology":null},{"url":"/paper/nuplanqa-a-large-scale-dataset-and-benchmark","slug":"nuplanqa-a-large-scale-dataset-and-benchmark","title":"NuPlanQA: A Large-Scale Dataset and Benchmark for Multi-View Driving Scene Understanding in Multi-Modal Large Language Models","date":"2025-03-17","arxiv_id":"2503.12772","repositories_listed":1,"syntology":null},{"url":"/paper/open3dvqa-a-benchmark-for-comprehensive","slug":"open3dvqa-a-benchmark-for-comprehensive","title":"Open3DVQA: A Benchmark for Comprehensive Spatial Reasoning with Multimodal Large Language Model in Open Space","date":"2025-03-14","arxiv_id":"2503.11094","repositories_listed":1,"syntology":null},{"url":"/paper/t2i-fineeval-fine-grained-compositional","slug":"t2i-fineeval-fine-grained-compositional","title":"T2I-FineEval: Fine-Grained Compositional Metric for Text-to-Image Evaluation","date":"2025-03-14","arxiv_id":"2503.11481","repositories_listed":1,"syntology":null},{"url":"/paper/drivelmm-o1-a-step-by-step-reasoning-dataset","slug":"drivelmm-o1-a-step-by-step-reasoning-dataset","title":"DriveLMM-o1: A Step-by-Step Reasoning Dataset and Large Multimodal Model for Driving Scenario Understanding","date":"2025-03-13","arxiv_id":"2503.10621","repositories_listed":1,"syntology":null},{"url":"/paper/kvq-boosting-video-quality-assessment-via","slug":"kvq-boosting-video-quality-assessment-via","title":"KVQ: Boosting Video Quality Assessment via Saliency-guided Local Perception","date":"2025-03-13","arxiv_id":"2503.10259","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":5,"n_ran_checked":5,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"9 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/kvq-boosting-video-quality-assessment-via#ran","syntology_url":"https://syntology.ai/paper/2503.10259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10259"}},"official":{"repos":["qyp2000/kvq"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":5,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/segagent-exploring-pixel-understanding","slug":"segagent-exploring-pixel-understanding","title":"SegAgent: Exploring Pixel Understanding Capabilities in MLLMs by Imitating Human Annotator Trajectories","date":"2025-03-11","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/segagent-exploring-pixel-understanding-1","slug":"segagent-exploring-pixel-understanding-1","title":"SegAgent: Exploring Pixel Understanding Capabilities in MLLMs by Imitating Human Annotator Trajectories","date":"2025-03-11","arxiv_id":"2503.08625","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/segagent-exploring-pixel-understanding-1#ran","syntology_url":"https://syntology.ai/paper/2503.08625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08625"}},"official":{"repos":["aim-uofa/SegAgent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/when-large-vision-language-model-meets-large","slug":"when-large-vision-language-model-meets-large","title":"When Large Vision-Language Model Meets Large Remote Sensing Imagery: Coarse-to-Fine Text-Guided Token Pruning","date":"2025-03-10","arxiv_id":"2503.07588","repositories_listed":1,"syntology":null},{"url":"/paper/next-token-is-enough-realistic-image-quality","slug":"next-token-is-enough-realistic-image-quality","title":"Next Token Is Enough: Realistic Image Quality and Aesthetic Scoring with Multimodal Large Language Model","date":"2025-03-08","arxiv_id":"2503.06141","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-vietnamese-vqa-through-curriculum","slug":"enhancing-vietnamese-vqa-through-curriculum","title":"Enhancing Vietnamese VQA through Curriculum Learning on Raw and Augmented Text Representations","date":"2025-03-05","arxiv_id":"2503.03285","repositories_listed":1,"syntology":null},{"url":"/paper/biod2c-a-dual-level-semantic-consistency","slug":"biod2c-a-dual-level-semantic-consistency","title":"BioD2C: A Dual-level Semantic Consistency Constraint Framework for Biomedical VQA","date":"2025-03-04","arxiv_id":"2503.02476","repositories_listed":1,"syntology":null},{"url":"/paper/watch-out-your-album-on-the-inadvertent","slug":"watch-out-your-album-on-the-inadvertent","title":"Watch Out Your Album! On the Inadvertent Privacy Memorization in Multi-Modal Large Language Models","date":"2025-03-03","arxiv_id":"2503.01208","repositories_listed":1,"syntology":null},{"url":"/paper/medhalltune-an-instruction-tuning-benchmark","slug":"medhalltune-an-instruction-tuning-benchmark","title":"MedHallTune: An Instruction-Tuning Benchmark for Mitigating Medical Hallucination in Vision-Language Models","date":"2025-02-28","arxiv_id":"2502.20780","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-score-alignment-learning-for","slug":"adaptive-score-alignment-learning-for","title":"Adaptive Score Alignment Learning for Continual Perceptual Quality Assessment of 360-Degree Videos in Virtual Reality","date":"2025-02-27","arxiv_id":"2502.19644","repositories_listed":1,"syntology":null},{"url":"/paper/omnialign-v-towards-enhanced-alignment-of","slug":"omnialign-v-towards-enhanced-alignment-of","title":"OmniAlign-V: Towards Enhanced Alignment of MLLMs with Human Preference","date":"2025-02-25","arxiv_id":"2502.18411","repositories_listed":1,"syntology":null},{"url":"/paper/multiscale-byte-language-models-a","slug":"multiscale-byte-language-models-a","title":"Multiscale Byte Language Models -- A Hierarchical Architecture for Causal Million-Length Sequence Modeling","date":"2025-02-20","arxiv_id":"2502.14553","repositories_listed":1,"syntology":null},{"url":"/paper/pitvqa-vector-matrix-low-rank-adaptation-for","slug":"pitvqa-vector-matrix-low-rank-adaptation-for","title":"PitVQA++: Vector Matrix-Low-Rank Adaptation for Open-Ended Visual Question Answering in Pituitary Surgery","date":"2025-02-19","arxiv_id":"2502.14149","repositories_listed":1,"syntology":null},{"url":"/paper/re-align-aligning-vision-language-models-via","slug":"re-align-aligning-vision-language-models-via","title":"Re-Align: Aligning Vision Language Models via Retrieval-Augmented Direct Preference Optimization","date":"2025-02-18","arxiv_id":"2502.13146","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/re-align-aligning-vision-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2502.13146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13146"}},"official":{"repos":["taco-group/re-align"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mmunlearner-reformulating-multimodal-machine","slug":"mmunlearner-reformulating-multimodal-machine","title":"MMUnlearner: Reformulating Multimodal Machine Unlearning in the Era of Multimodal Large Language Models","date":"2025-02-16","arxiv_id":"2502.11051","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mmunlearner-reformulating-multimodal-machine#ran","syntology_url":"https://syntology.ai/paper/2502.11051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11051"}},"official":{"repos":["z1zs/mmunlearner"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/user-vlm-360-personalized-vision-language","slug":"user-vlm-360-personalized-vision-language","title":"USER-VLM 360: Personalized Vision Language Models with User-aware Tuning for Social Human-Robot Interactions","date":"2025-02-15","arxiv_id":"2502.10636","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-potential-of-encoder-free","slug":"exploring-the-potential-of-encoder-free","title":"Exploring the Potential of Encoder-free Architectures in 3D LMMs","date":"2025-02-13","arxiv_id":"2502.09620","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-the-potential-of-encoder-free#ran","syntology_url":"https://syntology.ai/paper/2502.09620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.09620"}},"official":{"repos":["ivan-tang-3d/enel"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/clinkd-cross-modal-clinical-knowledge","slug":"clinkd-cross-modal-clinical-knowledge","title":"ClinKD: Cross-Modal Clinical Knowledge Distiller For Multi-Task Medical Images","date":"2025-02-09","arxiv_id":"2502.05928","repositories_listed":1,"syntology":null},{"url":"/paper/content-rich-aigc-video-quality-assessment","slug":"content-rich-aigc-video-quality-assessment","title":"Content-Rich AIGC Video Quality Assessment via Intricate Text Alignment and Motion-Aware Consistency","date":"2025-02-06","arxiv_id":"2502.04076","repositories_listed":1,"syntology":null},{"url":"/paper/no-images-no-problem-retaining-knowledge-in","slug":"no-images-no-problem-retaining-knowledge-in","title":"No Images, No Problem: Retaining Knowledge in Continual VQA with Questions-Only Memory","date":"2025-02-06","arxiv_id":"2502.04469","repositories_listed":1,"syntology":null},{"url":"/paper/variational-quantum-optimization-with","slug":"variational-quantum-optimization-with","title":"Variational Quantum Optimization with Continuous Bandits","date":"2025-02-06","arxiv_id":"2502.04021","repositories_listed":1,"syntology":null},{"url":"/paper/robust-llava-on-the-effectiveness-of-large","slug":"robust-llava-on-the-effectiveness-of-large","title":"Robust-LLaVA: On the Effectiveness of Large-Scale Robust Image Encoders for Multi-modal Large Language Models","date":"2025-02-03","arxiv_id":"2502.01576","repositories_listed":1,"syntology":null},{"url":"/paper/large-models-in-dialogue-for-active","slug":"large-models-in-dialogue-for-active","title":"Large Models in Dialogue for Active Perception and Anomaly Detection","date":"2025-01-27","arxiv_id":"2501.16300","repositories_listed":1,"syntology":null}],"record_sha256":"c4ed6df038341716b928edc0dc3b25e5eac4311984b1a679544cb759737c2485","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}