{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/6","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":22,"rows_per_page":100,"rows":[501,600],"of":2167,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering","prev":"/task/visual-question-answering/papers/5","next":"/task/visual-question-answering/papers/7","papers":[{"url":"/paper/subjective-and-objective-analysis-of-indian","slug":"subjective-and-objective-analysis-of-indian","title":"Subjective and Objective Analysis of Indian Social Media Video Quality","date":"2024-01-05","arxiv_id":"2401.02794","repositories_listed":1,"syntology":null},{"url":"/paper/artquest-countering-hidden-language-biases-in","slug":"artquest-countering-hidden-language-biases-in","title":"ArtQuest: Countering Hidden Language Biases in ArtVQA","date":"2024-01-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mining-fine-grained-image-text-alignment-for","slug":"mining-fine-grained-image-text-alignment-for","title":"Mining Fine-Grained Image-Text Alignment for Zero-Shot Captioning via Text-Only Training","date":"2024-01-04","arxiv_id":"2401.02347","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mining-fine-grained-image-text-alignment-for#ran","syntology_url":"https://syntology.ai/paper/2401.02347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02347"}},"official":{"repos":["artanic30/maccap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/q-align-teaching-lmms-for-visual-scoring-via","slug":"q-align-teaching-lmms-for-visual-scoring-via","title":"Q-Align: Teaching LMMs for Visual Scoring via Discrete Text-Defined Levels","date":"2023-12-28","arxiv_id":"2312.17090","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/q-align-teaching-lmms-for-visual-scoring-via#ran","syntology_url":"https://syntology.ai/paper/2312.17090","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.17090"}},"official":{"repos":["q-future/q-align"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-guided-semi-supervised-learning-for","slug":"knowledge-guided-semi-supervised-learning-for","title":"Knowledge Guided Semi-Supervised Learning for Quality Assessment of User Generated Videos","date":"2023-12-24","arxiv_id":"2312.15425","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-unified-multimodal-reasoning","slug":"towards-a-unified-multimodal-reasoning","title":"Towards a Unified Multimodal Reasoning Framework","date":"2023-12-22","arxiv_id":"2312.15021","repositories_listed":1,"syntology":null},{"url":"/paper/towards-more-faithful-natural-language","slug":"towards-more-faithful-natural-language","title":"Towards More Faithful Natural Language Explanation Using Multi-Level Contrastive Learning in VQA","date":"2023-12-21","arxiv_id":"2312.13594","repositories_listed":1,"syntology":null},{"url":"/paper/object-attribute-matters-in-visual-question","slug":"object-attribute-matters-in-visual-question","title":"Object Attribute Matters in Visual Question Answering","date":"2023-12-20","arxiv_id":"2401.09442","repositories_listed":1,"syntology":null},{"url":"/paper/earthvqa-towards-queryable-earth-via","slug":"earthvqa-towards-queryable-earth-via","title":"EarthVQA: Towards Queryable Earth via Relational Reasoning-Based Remote Sensing Visual Question Answering","date":"2023-12-19","arxiv_id":"2312.12222","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/earthvqa-towards-queryable-earth-via#ran","syntology_url":"https://syntology.ai/paper/2312.12222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12222"}},"official":{"repos":["Junjue-Wang/EarthVQA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vqa4cir-boosting-composed-image-retrieval","slug":"vqa4cir-boosting-composed-image-retrieval","title":"VQA4CIR: Boosting Composed Image Retrieval with Visual Question Answering","date":"2023-12-19","arxiv_id":"2312.12273","repositories_listed":1,"syntology":null},{"url":"/paper/haar-text-conditioned-generative-model-of-3d","slug":"haar-text-conditioned-generative-model-of-3d","title":"HAAR: Text-Conditioned Generative Model of 3D Strand-based Human Hairstyles","date":"2023-12-18","arxiv_id":"2312.11666","repositories_listed":1,"syntology":null},{"url":"/paper/m2conceptbase-a-fine-grained-aligned-multi","slug":"m2conceptbase-a-fine-grained-aligned-multi","title":"M^2ConceptBase: A Fine-Grained Aligned Concept-Centric Multimodal Knowledge Base","date":"2023-12-16","arxiv_id":"2312.10417","repositories_listed":1,"syntology":null},{"url":"/paper/vlap-efficient-video-language-alignment-via","slug":"vlap-efficient-video-language-alignment-via","title":"ViLA: Efficient Video-Language Alignment for Video Question Answering","date":"2023-12-13","arxiv_id":"2312.08367","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlap-efficient-video-language-alignment-via#ran","syntology_url":"https://syntology.ai/paper/2312.08367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08367"}},"official":{"repos":["xijun-cs/vila"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/genixer-empowering-multimodal-large-language","slug":"genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","arxiv_id":"2312.06731","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genixer-empowering-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.06731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06731"}},"official":{"repos":["zhaohengyuan1/genixer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/nuscenes-mqa-integrated-evaluation-of","slug":"nuscenes-mqa-integrated-evaluation-of","title":"NuScenes-MQA: Integrated Evaluation of Captions and QA for Autonomous Driving Datasets using Markup Annotations","date":"2023-12-11","arxiv_id":"2312.06352","repositories_listed":1,"syntology":null},{"url":"/paper/language-informed-visual-concept-learning","slug":"language-informed-visual-concept-learning","title":"Language-Informed Visual Concept Learning","date":"2023-12-06","arxiv_id":"2312.03587","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/language-informed-visual-concept-learning#ran","syntology_url":"https://syntology.ai/paper/2312.03587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03587"}},"official":{"repos":["sharonal10/langint"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-configure-good-in-context-sequence-for","slug":"how-to-configure-good-in-context-sequence-for","title":"How to Configure Good In-Context Sequence for Visual Question Answering","date":"2023-12-04","arxiv_id":"2312.01571","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-configure-good-in-context-sequence-for#ran","syntology_url":"https://syntology.ai/paper/2312.01571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01571"}},"official":{"repos":["garyjiajia/ofv2_icl_vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recursive-visual-programming","slug":"recursive-visual-programming","title":"Recursive Visual Programming","date":"2023-12-04","arxiv_id":"2312.02249","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/recursive-visual-programming#ran","syntology_url":"https://syntology.ai/paper/2312.02249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02249"}},"official":{"repos":["para-lost/rvp"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/debiasing-multimodal-models-via-causal","slug":"debiasing-multimodal-models-via-causal","title":"Debiasing Multimodal Models via Causal Information Minimization","date":"2023-11-28","arxiv_id":"2311.16941","repositories_listed":1,"syntology":null},{"url":"/paper/fully-authentic-visual-question-answering","slug":"fully-authentic-visual-question-answering","title":"Fully Authentic Visual Question Answering Dataset from Online Communities","date":"2023-11-27","arxiv_id":"2311.15562","repositories_listed":1,"syntology":null},{"url":"/paper/how-many-unicorns-are-in-this-image-a-safety","slug":"how-many-unicorns-are-in-this-image-a-safety","title":"How Many Unicorns Are in This Image? A Safety Evaluation Benchmark for Vision LLMs","date":"2023-11-27","arxiv_id":"2311.16101","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-many-unicorns-are-in-this-image-a-safety#ran","syntology_url":"https://syntology.ai/paper/2311.16101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16101"}},"official":{"repos":["ucsc-vlaa/vllm-safety-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-the-power-of-small-multimodal","slug":"boosting-the-power-of-small-multimodal","title":"Boosting the Power of Small Multimodal Reasoning Models to Match Larger Models with Self-Consistency Training","date":"2023-11-23","arxiv_id":"2311.14109","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-the-power-of-small-multimodal#ran","syntology_url":"https://syntology.ai/paper/2311.14109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.14109"}},"official":{"repos":["chengtan9907/mc-cot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/filling-the-image-information-gap-for-vqa","slug":"filling-the-image-information-gap-for-vqa","title":"Filling the Image Information Gap for VQA: Prompting Large Language Models to Proactively Ask Questions","date":"2023-11-20","arxiv_id":"2311.11598","repositories_listed":1,"syntology":null},{"url":"/paper/hidro-vqa-high-dynamic-range-oracle-for-video","slug":"hidro-vqa-high-dynamic-range-oracle-for-video","title":"HIDRO-VQA: High Dynamic Range Oracle for Video Quality Assessment","date":"2023-11-18","arxiv_id":"2311.11059","repositories_listed":1,"syntology":null},{"url":"/paper/attribute-diversity-determines-the","slug":"attribute-diversity-determines-the","title":"Attribute Diversity Determines the Systematicity Gap in VQA","date":"2023-11-15","arxiv_id":"2311.08695","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attribute-diversity-determines-the#ran","syntology_url":"https://syntology.ai/paper/2311.08695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08695"}},"official":{"repos":["ikb-a/systematicity-gap-in-vqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-zero-shot-visual-question-answering","slug":"improving-zero-shot-visual-question-answering","title":"Improving Zero-shot Visual Question Answering via Large Language Models with Reasoning Question Prompts","date":"2023-11-15","arxiv_id":"2311.09050","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-evaluation-of-gpt-4v-on","slug":"a-comprehensive-evaluation-of-gpt-4v-on","title":"A Comprehensive Evaluation of GPT-4V on Knowledge-Intensive Visual Question Answering","date":"2023-11-13","arxiv_id":"2311.07536","repositories_listed":1,"syntology":null},{"url":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and","slug":"sphinx-the-joint-mixing-of-weights-tasks-and","title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","date":"2023-11-13","arxiv_id":"2311.07575","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and#ran","syntology_url":"https://syntology.ai/paper/2311.07575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07575"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/monkey-image-resolution-and-text-label-are","slug":"monkey-image-resolution-and-text-label-are","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","date":"2023-11-11","arxiv_id":"2311.06607","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monkey-image-resolution-and-text-label-are#ran","syntology_url":"https://syntology.ai/paper/2311.06607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06607"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/analyzing-modular-approaches-for-visual","slug":"analyzing-modular-approaches-for-visual","title":"Analyzing Modular Approaches for Visual Question Decomposition","date":"2023-11-10","arxiv_id":"2311.06411","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/analyzing-modular-approaches-for-visual#ran","syntology_url":"https://syntology.ai/paper/2311.06411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06411"}},"official":{"repos":["brown-palm/visual-question-decomposition"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/zero-shot-translation-of-attention-patterns","slug":"zero-shot-translation-of-attention-patterns","title":"Zero-shot Translation of Attention Patterns in VQA Models to Natural Language","date":"2023-11-08","arxiv_id":"2311.05043","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-grounding-potential-of-vqa-oriented","slug":"exploring-grounding-potential-of-vqa-oriented","title":"GPT-4V-AD: Exploring Grounding Potential of VQA-oriented GPT-4V for Zero-shot Anomaly Detection","date":"2023-11-05","arxiv_id":"2311.02612","repositories_listed":1,"syntology":null},{"url":"/paper/language-guided-visual-question-answering","slug":"language-guided-visual-question-answering","title":"Language Guided Visual Question Answering: Elevate Your Multimodal Language Model Using Knowledge-Enriched Prompts","date":"2023-10-31","arxiv_id":"2310.20159","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-guided-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2310.20159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20159"}},"official":{"repos":["declare-lab/lg-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-task-and-weight-prioritization","slug":"dynamic-task-and-weight-prioritization","title":"Dynamic Task and Weight Prioritization Curriculum Learning for Multimodal Imagery","date":"2023-10-29","arxiv_id":"2310.19109","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-chatgpt-for-medical-applications","slug":"multimodal-chatgpt-for-medical-applications","title":"Multimodal ChatGPT for Medical Applications: an Experimental Study of GPT-4V","date":"2023-10-29","arxiv_id":"2310.19061","repositories_listed":1,"syntology":null},{"url":"/paper/viclevr-a-visual-reasoning-dataset-and-hybrid","slug":"viclevr-a-visual-reasoning-dataset-and-hybrid","title":"ViCLEVR: A Visual Reasoning Dataset and Hybrid Multimodal Fusion Model for Visual Question Answering in Vietnamese","date":"2023-10-27","arxiv_id":"2310.18046","repositories_listed":1,"syntology":null},{"url":"/paper/incorporating-probing-signals-into-multimodal","slug":"incorporating-probing-signals-into-multimodal","title":"Incorporating Probing Signals into Multimodal Machine Translation via Visual Question-Answering Pairs","date":"2023-10-26","arxiv_id":"2310.17133","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-are-temporal-and-causal","slug":"large-language-models-are-temporal-and-causal","title":"Large Language Models are Temporal and Causal Reasoners for Video Question Answering","date":"2023-10-24","arxiv_id":"2310.15747","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/large-language-models-are-temporal-and-causal#ran","syntology_url":"https://syntology.ai/paper/2310.15747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15747"}},"official":{"repos":["mlvlab/Flipped-VQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/visual-cropping-improves-zero-shot-question","slug":"visual-cropping-improves-zero-shot-question","title":"Towards Perceiving Small Visual Details in Zero-shot Visual Question Answering with Multimodal LLMs","date":"2023-10-24","arxiv_id":"2310.16033","repositories_listed":1,"syntology":null},{"url":"/paper/rsadapter-adapting-multimodal-models-for","slug":"rsadapter-adapting-multimodal-models-for","title":"RSAdapter: Adapting Multimodal Models for Remote Sensing Visual Question Answering","date":"2023-10-19","arxiv_id":"2310.13120","repositories_listed":1,"syntology":null},{"url":"/paper/unanswerable-visual-question-answering","slug":"unanswerable-visual-question-answering","title":"UNK-VQA: A Dataset and a Probe into the Abstention Ability of Multi-modal Large Models","date":"2023-10-17","arxiv_id":"2310.10942","repositories_listed":1,"syntology":null},{"url":"/paper/vlis-unimodal-language-models-guide","slug":"vlis-unimodal-language-models-guide","title":"VLIS: Unimodal Language Models Guide Multimodal Language Generation","date":"2023-10-15","arxiv_id":"2310.09767","repositories_listed":1,"syntology":null},{"url":"/paper/pali-3-vision-language-models-smaller-faster","slug":"pali-3-vision-language-models-smaller-faster","title":"PaLI-3 Vision Language Models: Smaller, Faster, Stronger","date":"2023-10-13","arxiv_id":"2310.09199","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pali-3-vision-language-models-smaller-faster#ran","syntology_url":"https://syntology.ai/paper/2310.09199","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09199"}},"official":null}},{"url":"/paper/open-set-knowledge-based-visual-question","slug":"open-set-knowledge-based-visual-question","title":"Open-Set Knowledge-Based Visual Question Answering with Inference Paths","date":"2023-10-12","arxiv_id":"2310.08148","repositories_listed":1,"syntology":null},{"url":"/paper/what-if-the-tv-was-off-examining","slug":"what-if-the-tv-was-off-examining","title":"What If the TV Was Off? Examining Counterfactual Reasoning Abilities of Multi-modal Language Models","date":"2023-10-10","arxiv_id":"2310.06627","repositories_listed":1,"syntology":null},{"url":"/paper/rephrase-augment-reason-visual-grounding-of","slug":"rephrase-augment-reason-visual-grounding-of","title":"Rephrase, Augment, Reason: Visual Grounding of Questions for Vision-Language Models","date":"2023-10-09","arxiv_id":"2310.05861","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rephrase-augment-reason-visual-grounding-of#ran","syntology_url":"https://syntology.ai/paper/2310.05861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05861"}},"official":{"repos":["archiki/repare"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/navigating-cultural-chasms-exploring-and","slug":"navigating-cultural-chasms-exploring-and","title":"Navigating Cultural Chasms: Exploring and Unlocking the Cultural POV of Text-To-Image Models","date":"2023-10-03","arxiv_id":"2310.01929","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-task-performance-evaluating-and","slug":"beyond-task-performance-evaluating-and","title":"Beyond Task Performance: Evaluating and Reducing the Flaws of Large Multimodal Models with In-Context Learning","date":"2023-10-01","arxiv_id":"2310.00647","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-task-performance-evaluating-and#ran","syntology_url":"https://syntology.ai/paper/2310.00647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00647"}},"official":{"repos":["mshukor/EvALign-ICL"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fine-grained-late-interaction-multi-modal-1","slug":"fine-grained-late-interaction-multi-modal-1","title":"Fine-grained Late-interaction Multi-modal Retrieval for Retrieval Augmented Visual Question Answering","date":"2023-09-29","arxiv_id":"2309.17133","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-grained-late-interaction-multi-modal-1#ran","syntology_url":"https://syntology.ai/paper/2309.17133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17133"}},"official":{"repos":["linweizhedragon/retrieval-augmented-visual-question-answering"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/elip-efficient-language-image-pre-training","slug":"elip-efficient-language-image-pre-training","title":"ELIP: Efficient Language-Image Pre-training with Fewer Vision Tokens","date":"2023-09-28","arxiv_id":"2309.16738","repositories_listed":1,"syntology":null},{"url":"/paper/vulnerabilities-in-video-quality-assessment-1","slug":"vulnerabilities-in-video-quality-assessment-1","title":"Vulnerabilities in Video Quality Assessment Models: The Challenge of Adversarial Attacks","date":"2023-09-24","arxiv_id":"2309.13609","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vulnerabilities-in-video-quality-assessment-1#ran","syntology_url":"https://syntology.ai/paper/2309.13609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.13609"}},"official":{"repos":["gzhu-dvl/attackvqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/implicit-differentiable-outlier-detection","slug":"implicit-differentiable-outlier-detection","title":"Implicit Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-09-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/d3-data-diversity-design-for-systematic","slug":"d3-data-diversity-design-for-systematic","title":"D3: Data Diversity Design for Systematic Generalization in Visual Question Answering","date":"2023-09-15","arxiv_id":"2309.08798","repositories_listed":1,"syntology":null},{"url":"/paper/can-i-trust-your-answer-visually-grounded","slug":"can-i-trust-your-answer-visually-grounded","title":"Can I Trust Your Answer? Visually Grounded Video Question Answering","date":"2023-09-04","arxiv_id":"2309.01327","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/can-i-trust-your-answer-visually-grounded#ran","syntology_url":"https://syntology.ai/paper/2309.01327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01327"}},"official":{"repos":["doc-doc/next-gqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/separate-and-locate-rethink-the-text-in-text","slug":"separate-and-locate-rethink-the-text-in-text","title":"Separate and Locate: Rethink the Text in Text-based Visual Question Answering","date":"2023-08-31","arxiv_id":"2308.16383","repositories_listed":1,"syntology":null},{"url":"/paper/vqa-therapy-exploring-answer-differences-by","slug":"vqa-therapy-exploring-answer-differences-by","title":"VQA Therapy: Exploring Answer Differences by Visually Grounding Answers","date":"2023-08-21","arxiv_id":"2308.11662","repositories_listed":1,"syntology":null},{"url":"/paper/bliva-a-simple-multimodal-llm-for-better","slug":"bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","arxiv_id":"2308.09936","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bliva-a-simple-multimodal-llm-for-better#ran","syntology_url":"https://syntology.ai/paper/2308.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09936"}},"official":{"repos":["mlpc-ucsd/bliva"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/open-vocabulary-video-question-answering-a","slug":"open-vocabulary-video-question-answering-a","title":"Open-vocabulary Video Question Answering: A New Benchmark for Evaluating the Generalizability of Video Question Answering Models","date":"2023-08-18","arxiv_id":"2308.09363","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/open-vocabulary-video-question-answering-a#ran","syntology_url":"https://syntology.ai/paper/2308.09363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09363"}},"official":{"repos":["mlvlab/ovqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-pet-vision-and-language-parameter","slug":"vl-pet-vision-and-language-parameter","title":"VL-PET: Vision-and-Language Parameter-Efficient Tuning via Granularity Control","date":"2023-08-18","arxiv_id":"2308.09804","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vl-pet-vision-and-language-parameter#ran","syntology_url":"https://syntology.ai/paper/2308.09804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09804"}},"official":{"repos":["henryhzy/vl-pet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tech-text-guided-reconstruction-of-lifelike","slug":"tech-text-guided-reconstruction-of-lifelike","title":"TeCH: Text-guided Reconstruction of Lifelike Clothed Humans","date":"2023-08-16","arxiv_id":"2308.08545","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tech-text-guided-reconstruction-of-lifelike#ran","syntology_url":"https://syntology.ai/paper/2308.08545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08545"}},"official":{"repos":["huangyangyi/tech"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ugc-quality-assessment-exploring-the-impact","slug":"ugc-quality-assessment-exploring-the-impact","title":"UGC Quality Assessment: Exploring the Impact of Saliency in Deep Feature-Based Quality Assessment","date":"2023-08-13","arxiv_id":"2308.06853","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-and-preventing-hallucinations-in","slug":"detecting-and-preventing-hallucinations-in","title":"Detecting and Preventing Hallucinations in Large Vision Language Models","date":"2023-08-11","arxiv_id":"2308.06394","repositories_listed":1,"syntology":null},{"url":"/paper/stablevqa-a-deep-no-reference-quality","slug":"stablevqa-a-deep-no-reference-quality","title":"StableVQA: A Deep No-Reference Quality Assessment Model for Video Stability","date":"2023-08-09","arxiv_id":"2308.04904","repositories_listed":1,"syntology":null},{"url":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn","slug":"scigraphqa-a-large-scale-synthetic-multi-turn","title":"SciGraphQA: A Large-Scale Synthetic Multi-Turn Question-Answering Dataset for Scientific Graphs","date":"2023-08-07","arxiv_id":"2308.03349","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn#ran","syntology_url":"https://syntology.ai/paper/2308.03349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03349"}},"official":{"repos":["findalexli/SciGraphQA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/context-vqa-towards-context-aware-and","slug":"context-vqa-towards-context-aware-and","title":"Context-VQA: Towards Context-Aware and Purposeful Visual Question Answering","date":"2023-07-28","arxiv_id":"2307.15745","repositories_listed":1,"syntology":null},{"url":"/paper/analysis-of-video-quality-datasets-via-design","slug":"analysis-of-video-quality-datasets-via-design","title":"Analysis of Video Quality Datasets via Design of Minimalistic Video Quality Models","date":"2023-07-26","arxiv_id":"2307.13981","repositories_listed":1,"syntology":null},{"url":"/paper/expert-knowledge-aware-image-difference-graph","slug":"expert-knowledge-aware-image-difference-graph","title":"Expert Knowledge-Aware Image Difference Graph Representation Learning for Difference-Aware Medical Visual Question Answering","date":"2023-07-22","arxiv_id":"2307.11986","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expert-knowledge-aware-image-difference-graph#ran","syntology_url":"https://syntology.ai/paper/2307.11986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11986"}},"official":{"repos":["holipori/mimic-diff-vqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/explaining-autonomous-driving-actions-with","slug":"explaining-autonomous-driving-actions-with","title":"Explaining Autonomous Driving Actions with Visual Question Answering","date":"2023-07-19","arxiv_id":"2307.10408","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-performance-analysis-on-pre-trained","slug":"towards-a-performance-analysis-on-pre-trained","title":"Towards a performance analysis on pre-trained Visual Question Answering models for autonomous driving","date":"2023-07-18","arxiv_id":"2307.09329","repositories_listed":1,"syntology":null},{"url":"/paper/co-attention-gated-vision-language-embedding","slug":"co-attention-gated-vision-language-embedding","title":"CAT-ViL: Co-Attention Gated Vision-Language Embedding for Visual Question Localized-Answering in Robotic Surgery","date":"2023-07-11","arxiv_id":"2307.05182","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-attention-gated-vision-language-embedding#ran","syntology_url":"https://syntology.ai/paper/2307.05182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05182"}},"official":{"repos":["longbai1006/cat-vil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rad-restruct-a-novel-vqa-benchmark-and-method","slug":"rad-restruct-a-novel-vqa-benchmark-and-method","title":"Rad-ReStruct: A Novel VQA Benchmark and Method for Structured Radiology Reporting","date":"2023-07-11","arxiv_id":"2307.05766","repositories_listed":1,"syntology":null},{"url":"/paper/subjective-and-objective-audio-visual-quality","slug":"subjective-and-objective-audio-visual-quality","title":"Subjective and Objective Audio-Visual Quality Assessment for User Generated Content","date":"2023-07-10","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/localized-questions-in-medical-visual","slug":"localized-questions-in-medical-visual","title":"Localized Questions in Medical Visual Question Answering","date":"2023-07-03","arxiv_id":"2307.01067","repositories_listed":1,"syntology":null},{"url":"/paper/unifine-a-unified-and-fine-grained-approach","slug":"unifine-a-unified-and-fine-grained-approach","title":"UniFine: A Unified and Fine-grained Approach for Zero-shot Vision-Language Understanding","date":"2023-07-03","arxiv_id":"2307.00862","repositories_listed":1,"syntology":null},{"url":"/paper/lightweight-recurrent-cross-modal-encoder-for","slug":"lightweight-recurrent-cross-modal-encoder-for","title":"Lightweight Recurrent Cross-modal Encoder for Video Question Answering","date":"2023-06-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-prompt-retrieval-for-generative","slug":"multimodal-prompt-retrieval-for-generative","title":"Multimodal Prompt Retrieval for Generative Visual Question Answering","date":"2023-06-30","arxiv_id":"2306.17675","repositories_listed":1,"syntology":null},{"url":"/paper/answer-mining-from-a-pool-of-images-towards","slug":"answer-mining-from-a-pool-of-images-towards","title":"Answer Mining from a Pool of Images: Towards Retrieval-Based Visual Question Answering","date":"2023-06-29","arxiv_id":"2306.16713","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/answer-mining-from-a-pool-of-images-towards#ran","syntology_url":"https://syntology.ai/paper/2306.16713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16713"}},"official":{"repos":["Abhiram4572/mi_bart"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-training-multi-modal-dense-retrievers-for","slug":"pre-training-multi-modal-dense-retrievers-for","title":"Pre-Training Multi-Modal Dense Retrievers for Outside-Knowledge Visual Question Answering","date":"2023-06-28","arxiv_id":"2306.16478","repositories_listed":1,"syntology":null},{"url":"/paper/shikra-unleashing-multimodal-llm-s","slug":"shikra-unleashing-multimodal-llm-s","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","date":"2023-06-27","arxiv_id":"2306.15195","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shikra-unleashing-multimodal-llm-s#ran","syntology_url":"https://syntology.ai/paper/2306.15195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15195"}},"official":{"repos":["shikras/shikra"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/funqa-towards-surprising-video-comprehension","slug":"funqa-towards-surprising-video-comprehension","title":"FunQA: Towards Surprising Video Comprehension","date":"2023-06-26","arxiv_id":"2306.14899","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/funqa-towards-surprising-video-comprehension#ran","syntology_url":"https://syntology.ai/paper/2306.14899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14899"}},"official":{"repos":["jingkang50/funqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/starvqa-co-training-space-time-attention-for","slug":"starvqa-co-training-space-time-attention-for","title":"StarVQA+: Co-training Space-Time Attention for Video Quality Assessment","date":"2023-06-21","arxiv_id":"2306.12298","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-prompting-techniques-for-zero","slug":"investigating-prompting-techniques-for-zero","title":"Investigating Prompting Techniques for Zero- and Few-Shot Visual Question Answering","date":"2023-06-16","arxiv_id":"2306.09996","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/investigating-prompting-techniques-for-zero#ran","syntology_url":"https://syntology.ai/paper/2306.09996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09996"}},"official":{"repos":["rabiulcste/vqazero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cosa-concatenated-sample-pretrained-vision","slug":"cosa-concatenated-sample-pretrained-vision","title":"COSA: Concatenated Sample Pretrained Vision-Language Foundation Model","date":"2023-06-15","arxiv_id":"2306.09085","repositories_listed":1,"syntology":null},{"url":"/paper/encyclopedic-vqa-visual-questions-about","slug":"encyclopedic-vqa-visual-questions-about","title":"Encyclopedic VQA: Visual questions about detailed properties of fine-grained categories","date":"2023-06-15","arxiv_id":"2306.09224","repositories_listed":1,"syntology":null},{"url":"/paper/improving-selective-visual-question-answering-1","slug":"improving-selective-visual-question-answering-1","title":"Improving Selective Visual Question Answering by Learning from Your Peers","date":"2023-06-14","arxiv_id":"2306.08751","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-neural-probabilistic-answer-set","slug":"scalable-neural-probabilistic-answer-set","title":"Scalable Neural-Probabilistic Answer Set Programming","date":"2023-06-14","arxiv_id":"2306.08397","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-neural-probabilistic-answer-set#ran","syntology_url":"https://syntology.ai/paper/2306.08397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08397"}},"official":{"repos":["ml-research/slash"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/modular-visual-question-answering-via-code","slug":"modular-visual-question-answering-via-code","title":"Modular Visual Question Answering via Code Generation","date":"2023-06-08","arxiv_id":"2306.05392","repositories_listed":1,"syntology":null},{"url":"/paper/rewarded-soups-towards-pareto-optimal-1","slug":"rewarded-soups-towards-pareto-optimal-1","title":"Rewarded soups: towards Pareto-optimal alignment by interpolating weights fine-tuned on diverse rewards","date":"2023-06-07","arxiv_id":"2306.04488","repositories_listed":1,"syntology":null},{"url":"/paper/q-how-to-specialize-large-vision-language-1","slug":"q-how-to-specialize-large-vision-language-1","title":"Q: How to Specialize Large Vision-Language Models to Data-Scarce VQA Tasks? A: Self-Train on Unlabeled Images!","date":"2023-06-06","arxiv_id":"2306.03932","repositories_listed":1,"syntology":null},{"url":"/paper/docformerv2-local-features-for-document","slug":"docformerv2-local-features-for-document","title":"DocFormerv2: Local Features for Document Understanding","date":"2023-06-02","arxiv_id":"2306.01733","repositories_listed":1,"syntology":null},{"url":"/paper/visualgptscore-visio-linguistic-reasoning","slug":"visualgptscore-visio-linguistic-reasoning","title":"Revisiting the Role of Language Priors in Vision-Language Models","date":"2023-06-02","arxiv_id":"2306.01879","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-knowledge-retrieval-with-multi","slug":"end-to-end-knowledge-retrieval-with-multi","title":"End-to-end Knowledge Retrieval with Multi-modal Queries","date":"2023-06-01","arxiv_id":"2306.00424","repositories_listed":1,"syntology":null},{"url":"/paper/havqa-a-dataset-for-visual-question-answering","slug":"havqa-a-dataset-for-visual-question-answering","title":"HaVQA: A Dataset for Visual Question Answering and Multimodal Research in Hausa Language","date":"2023-05-28","arxiv_id":"2305.17690","repositories_listed":1,"syntology":null},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/modularized-zero-shot-vqa-with-pre-trained","slug":"modularized-zero-shot-vqa-with-pre-trained","title":"Modularized Zero-shot VQA with Pre-trained Models","date":"2023-05-27","arxiv_id":"2305.17369","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-visual-question-answering-with","slug":"zero-shot-visual-question-answering-with","title":"Zero-shot Visual Question Answering with Language Model Feedback","date":"2023-05-26","arxiv_id":"2305.17006","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-faithful-and-plausible-visual","slug":"measuring-faithful-and-plausible-visual","title":"Measuring Faithful and Plausible Visual Grounding in VQA","date":"2023-05-24","arxiv_id":"2305.15015","repositories_listed":1,"syntology":null},{"url":"/paper/nuscenes-qa-a-multi-modal-visual-question","slug":"nuscenes-qa-a-multi-modal-visual-question","title":"NuScenes-QA: A Multi-modal Visual Question Answering Benchmark for Autonomous Driving Scenario","date":"2023-05-24","arxiv_id":"2305.14836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nuscenes-qa-a-multi-modal-visual-question#ran","syntology_url":"https://syntology.ai/paper/2305.14836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14836"}},"official":{"repos":["qiantianwen/nuscenes-qa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-explainable-in-the-wild-video-quality","slug":"towards-explainable-in-the-wild-video-quality","title":"Towards Explainable In-the-Wild Video Quality Assessment: A Database and a Language-Prompted Approach","date":"2023-05-22","arxiv_id":"2305.12726","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-vision-language-pre-training-with","slug":"enhancing-vision-language-pre-training-with","title":"Enhancing Vision-Language Pre-Training with Jointly Learned Questioner and Dense Captioner","date":"2023-05-19","arxiv_id":"2305.11769","repositories_listed":1,"syntology":null}],"record_sha256":"35c25782f314b0b9d990e7145befee83324de018c3cec0851a25c986fa1cedd7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}