{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering-1/papers/7","list_of":"/task/visual-question-answering-1","task":"Visual Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":22,"rows_per_page":100,"rows":[601,700],"of":2177,"counts":{"archive_papers_tagged":2177,"with_a_code_link":1042,"where_syntology_ran_a_sample":378,"not_listed_spam_title":0,"listed":2177,"listed_where_code_ran":378,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":308,"every_run_a_failure_of_syntologys_instrument":70,"listed_with_a_run_with_no_instrument_failure":308,"listed_every_run_a_failure_of_syntologys_instrument":70,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering-1","prev":"/task/visual-question-answering-1/papers/6","next":"/task/visual-question-answering-1/papers/8","papers":[{"url":"/paper/language-informed-visual-concept-learning","slug":"language-informed-visual-concept-learning","title":"Language-Informed Visual Concept Learning","date":"2023-12-06","arxiv_id":"2312.03587","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/language-informed-visual-concept-learning#ran","syntology_url":"https://syntology.ai/paper/2312.03587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03587"}},"official":{"repos":["sharonal10/langint"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/onellm-one-framework-to-align-all-modalities","slug":"onellm-one-framework-to-align-all-modalities","title":"OneLLM: One Framework to Align All Modalities with Language","date":"2023-12-06","arxiv_id":"2312.03700","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/onellm-one-framework-to-align-all-modalities#ran","syntology_url":"https://syntology.ai/paper/2312.03700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03700"}},"official":{"repos":["csuhan/onellm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/good-questions-help-zero-shot-image-reasoning","slug":"good-questions-help-zero-shot-image-reasoning","title":"Good Questions Help Zero-Shot Image Reasoning","date":"2023-12-04","arxiv_id":"2312.01598","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-configure-good-in-context-sequence-for","slug":"how-to-configure-good-in-context-sequence-for","title":"How to Configure Good In-Context Sequence for Visual Question Answering","date":"2023-12-04","arxiv_id":"2312.01571","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-configure-good-in-context-sequence-for#ran","syntology_url":"https://syntology.ai/paper/2312.01571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01571"}},"official":{"repos":["garyjiajia/ofv2_icl_vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recursive-visual-programming","slug":"recursive-visual-programming","title":"Recursive Visual Programming","date":"2023-12-04","arxiv_id":"2312.02249","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/recursive-visual-programming#ran","syntology_url":"https://syntology.ai/paper/2312.02249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02249"}},"official":{"repos":["para-lost/rvp"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/plug-and-play-dense-label-free-extraction-of","slug":"plug-and-play-dense-label-free-extraction-of","title":"Emergent Open-Vocabulary Semantic Segmentation from Off-the-shelf Vision-Language Models","date":"2023-11-28","arxiv_id":"2311.17095","repositories_listed":1,"syntology":null},{"url":"/paper/can-vision-language-models-think-from-a-first","slug":"can-vision-language-models-think-from-a-first","title":"EgoThink: Evaluating First-Person Perspective Thinking Capability of Vision-Language Models","date":"2023-11-27","arxiv_id":"2311.15596","repositories_listed":1,"syntology":null},{"url":"/paper/fully-authentic-visual-question-answering","slug":"fully-authentic-visual-question-answering","title":"Fully Authentic Visual Question Answering Dataset from Online Communities","date":"2023-11-27","arxiv_id":"2311.15562","repositories_listed":1,"syntology":null},{"url":"/paper/llmga-multimodal-large-language-model-based","slug":"llmga-multimodal-large-language-model-based","title":"LLMGA: Multimodal Large Language Model based Generation Assistant","date":"2023-11-27","arxiv_id":"2311.16500","repositories_listed":1,"syntology":null},{"url":"/paper/geochat-grounded-large-vision-language-model","slug":"geochat-grounded-large-vision-language-model","title":"GeoChat: Grounded Large Vision-Language Model for Remote Sensing","date":"2023-11-24","arxiv_id":"2311.15826","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/geochat-grounded-large-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2311.15826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15826"}},"official":{"repos":["mbzuai-oryx/geochat"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/sharegpt4v-improving-large-multi-modal-models","slug":"sharegpt4v-improving-large-multi-modal-models","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","date":"2023-11-21","arxiv_id":"2311.12793","repositories_listed":1,"syntology":null},{"url":"/paper/filling-the-image-information-gap-for-vqa","slug":"filling-the-image-information-gap-for-vqa","title":"Filling the Image Information Gap for VQA: Prompting Large Language Models to Proactively Ask Questions","date":"2023-11-20","arxiv_id":"2311.11598","repositories_listed":1,"syntology":null},{"url":"/paper/attribute-diversity-determines-the","slug":"attribute-diversity-determines-the","title":"Attribute Diversity Determines the Systematicity Gap in VQA","date":"2023-11-15","arxiv_id":"2311.08695","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attribute-diversity-determines-the#ran","syntology_url":"https://syntology.ai/paper/2311.08695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08695"}},"official":{"repos":["ikb-a/systematicity-gap-in-vqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-zero-shot-visual-question-answering","slug":"improving-zero-shot-visual-question-answering","title":"Improving Zero-shot Visual Question Answering via Large Language Models with Reasoning Question Prompts","date":"2023-11-15","arxiv_id":"2311.09050","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-evaluation-of-gpt-4v-on","slug":"a-comprehensive-evaluation-of-gpt-4v-on","title":"A Comprehensive Evaluation of GPT-4V on Knowledge-Intensive Visual Question Answering","date":"2023-11-13","arxiv_id":"2311.07536","repositories_listed":1,"syntology":null},{"url":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and","slug":"sphinx-the-joint-mixing-of-weights-tasks-and","title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","date":"2023-11-13","arxiv_id":"2311.07575","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and#ran","syntology_url":"https://syntology.ai/paper/2311.07575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07575"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/volcano-mitigating-multimodal-hallucination","slug":"volcano-mitigating-multimodal-hallucination","title":"Volcano: Mitigating Multimodal Hallucination through Self-Feedback Guided Revision","date":"2023-11-13","arxiv_id":"2311.07362","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/volcano-mitigating-multimodal-hallucination#ran","syntology_url":"https://syntology.ai/paper/2311.07362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07362"}},"official":{"repos":["kaistai/volcano"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/monkey-image-resolution-and-text-label-are","slug":"monkey-image-resolution-and-text-label-are","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","date":"2023-11-11","arxiv_id":"2311.06607","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monkey-image-resolution-and-text-label-are#ran","syntology_url":"https://syntology.ai/paper/2311.06607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06607"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-plus-learning-to-use-tools-for-creating","slug":"llava-plus-learning-to-use-tools-for-creating","title":"LLaVA-Plus: Learning to Use Tools for Creating Multimodal Agents","date":"2023-11-09","arxiv_id":"2311.05437","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-plus-learning-to-use-tools-for-creating#ran","syntology_url":"https://syntology.ai/paper/2311.05437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05437"}},"official":{"repos":["LLaVA-VL/LLaVA-Plus-Codebase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/genome-generative-neuro-symbolic-visual","slug":"genome-generative-neuro-symbolic-visual","title":"GENOME: GenerativE Neuro-symbOlic visual reasoning by growing and reusing ModulEs","date":"2023-11-08","arxiv_id":"2311.04901","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/genome-generative-neuro-symbolic-visual#ran","syntology_url":"https://syntology.ai/paper/2311.04901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04901"}},"official":null}},{"url":"/paper/zero-shot-translation-of-attention-patterns","slug":"zero-shot-translation-of-attention-patterns","title":"Zero-shot Translation of Attention Patterns in VQA Models to Natural Language","date":"2023-11-08","arxiv_id":"2311.05043","repositories_listed":1,"syntology":null},{"url":"/paper/otterhd-a-high-resolution-multi-modality","slug":"otterhd-a-high-resolution-multi-modality","title":"OtterHD: A High-Resolution Multi-modality Model","date":"2023-11-07","arxiv_id":"2311.04219","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/otterhd-a-high-resolution-multi-modality#ran","syntology_url":"https://syntology.ai/paper/2311.04219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04219"}},"official":{"repos":["luodian/otter"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-grounding-potential-of-vqa-oriented","slug":"exploring-grounding-potential-of-vqa-oriented","title":"GPT-4V-AD: Exploring Grounding Potential of VQA-oriented GPT-4V for Zero-shot Anomaly Detection","date":"2023-11-05","arxiv_id":"2311.02612","repositories_listed":1,"syntology":null},{"url":"/paper/language-guided-visual-question-answering","slug":"language-guided-visual-question-answering","title":"Language Guided Visual Question Answering: Elevate Your Multimodal Language Model Using Knowledge-Enriched Prompts","date":"2023-10-31","arxiv_id":"2310.20159","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-guided-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2310.20159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20159"}},"official":{"repos":["declare-lab/lg-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/making-large-language-models-better-data","slug":"making-large-language-models-better-data","title":"Making Large Language Models Better Data Creators","date":"2023-10-31","arxiv_id":"2310.20111","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-task-and-weight-prioritization","slug":"dynamic-task-and-weight-prioritization","title":"Dynamic Task and Weight Prioritization Curriculum Learning for Multimodal Imagery","date":"2023-10-29","arxiv_id":"2310.19109","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-follow-object-centric-image","slug":"learning-to-follow-object-centric-image","title":"Learning to Follow Object-Centric Image Editing Instructions Faithfully","date":"2023-10-29","arxiv_id":"2310.19145","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-chatgpt-for-medical-applications","slug":"multimodal-chatgpt-for-medical-applications","title":"Multimodal ChatGPT for Medical Applications: an Experimental Study of GPT-4V","date":"2023-10-29","arxiv_id":"2310.19061","repositories_listed":1,"syntology":null},{"url":"/paper/viclevr-a-visual-reasoning-dataset-and-hybrid","slug":"viclevr-a-visual-reasoning-dataset-and-hybrid","title":"ViCLEVR: A Visual Reasoning Dataset and Hybrid Multimodal Fusion Model for Visual Question Answering in Vietnamese","date":"2023-10-27","arxiv_id":"2310.18046","repositories_listed":1,"syntology":null},{"url":"/paper/antifakeprompt-prompt-tuned-vision-language","slug":"antifakeprompt-prompt-tuned-vision-language","title":"AntifakePrompt: Prompt-Tuned Vision-Language Models are Fake Image Detectors","date":"2023-10-26","arxiv_id":"2310.17419","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/antifakeprompt-prompt-tuned-vision-language#ran","syntology_url":"https://syntology.ai/paper/2310.17419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17419"}},"official":{"repos":["nctu-eva-lab/antifakeprompt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/incorporating-probing-signals-into-multimodal","slug":"incorporating-probing-signals-into-multimodal","title":"Incorporating Probing Signals into Multimodal Machine Translation via Visual Question-Answering Pairs","date":"2023-10-26","arxiv_id":"2310.17133","repositories_listed":1,"syntology":null},{"url":"/paper/visual-cropping-improves-zero-shot-question","slug":"visual-cropping-improves-zero-shot-question","title":"Towards Perceiving Small Visual Details in Zero-shot Visual Question Answering with Multimodal LLMs","date":"2023-10-24","arxiv_id":"2310.16033","repositories_listed":1,"syntology":null},{"url":"/paper/rsadapter-adapting-multimodal-models-for","slug":"rsadapter-adapting-multimodal-models-for","title":"RSAdapter: Adapting Multimodal Models for Remote Sensing Visual Question Answering","date":"2023-10-19","arxiv_id":"2310.13120","repositories_listed":1,"syntology":null},{"url":"/paper/unanswerable-visual-question-answering","slug":"unanswerable-visual-question-answering","title":"UNK-VQA: A Dataset and a Probe into the Abstention Ability of Multi-modal Large Models","date":"2023-10-17","arxiv_id":"2310.10942","repositories_listed":1,"syntology":null},{"url":"/paper/from-clip-to-dino-visual-encoders-shout-in","slug":"from-clip-to-dino-visual-encoders-shout-in","title":"From CLIP to DINO: Visual Encoders Shout in Multi-modal Large Language Models","date":"2023-10-13","arxiv_id":"2310.08825","repositories_listed":1,"syntology":null},{"url":"/paper/open-set-knowledge-based-visual-question","slug":"open-set-knowledge-based-visual-question","title":"Open-Set Knowledge-Based Visual Question Answering with Inference Paths","date":"2023-10-12","arxiv_id":"2310.08148","repositories_listed":1,"syntology":null},{"url":"/paper/rephrase-augment-reason-visual-grounding-of","slug":"rephrase-augment-reason-visual-grounding-of","title":"Rephrase, Augment, Reason: Visual Grounding of Questions for Vision-Language Models","date":"2023-10-09","arxiv_id":"2310.05861","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rephrase-augment-reason-visual-grounding-of#ran","syntology_url":"https://syntology.ai/paper/2310.05861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05861"}},"official":{"repos":["archiki/repare"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mathvista-evaluating-mathematical-reasoning","slug":"mathvista-evaluating-mathematical-reasoning","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","date":"2023-10-03","arxiv_id":"2310.02255","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathvista-evaluating-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.02255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02255"}},"official":null}},{"url":"/paper/fine-grained-late-interaction-multi-modal-1","slug":"fine-grained-late-interaction-multi-modal-1","title":"Fine-grained Late-interaction Multi-modal Retrieval for Retrieval Augmented Visual Question Answering","date":"2023-09-29","arxiv_id":"2309.17133","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-grained-late-interaction-multi-modal-1#ran","syntology_url":"https://syntology.ai/paper/2309.17133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17133"}},"official":{"repos":["linweizhedragon/retrieval-augmented-visual-question-answering"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/toloka-visual-question-answering-benchmark","slug":"toloka-visual-question-answering-benchmark","title":"Toloka Visual Question Answering Benchmark","date":"2023-09-28","arxiv_id":"2309.16511","repositories_listed":1,"syntology":null},{"url":"/paper/vdc-versatile-data-cleanser-for-detecting","slug":"vdc-versatile-data-cleanser-for-detecting","title":"VDC: Versatile Data Cleanser based on Visual-Linguistic Inconsistency by Multimodal Large Language Models","date":"2023-09-28","arxiv_id":"2309.16211","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":4,"n_ran_checked":5,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/vdc-versatile-data-cleanser-for-detecting#ran","syntology_url":"https://syntology.ai/paper/2309.16211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16211"}},"official":{"repos":["zihao-ai/vdc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/dreamllm-synergistic-multimodal-comprehension","slug":"dreamllm-synergistic-multimodal-comprehension","title":"DreamLLM: Synergistic Multimodal Comprehension and Creation","date":"2023-09-20","arxiv_id":"2309.11499","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dreamllm-synergistic-multimodal-comprehension#ran","syntology_url":"https://syntology.ai/paper/2309.11499","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11499"}},"official":{"repos":["RunpeiDong/DreamLLM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/an-empirical-study-of-scaling-instruct-tuned","slug":"an-empirical-study-of-scaling-instruct-tuned","title":"An Empirical Study of Scaling Instruct-Tuned Large Multimodal Models","date":"2023-09-18","arxiv_id":"2309.09958","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-of-scaling-instruct-tuned#ran","syntology_url":"https://syntology.ai/paper/2309.09958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.09958"}},"official":{"repos":["haotian-liu/LLaVA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/d3-data-diversity-design-for-systematic","slug":"d3-data-diversity-design-for-systematic","title":"D3: Data Diversity Design for Systematic Generalization in Visual Question Answering","date":"2023-09-15","arxiv_id":"2309.08798","repositories_listed":1,"syntology":null},{"url":"/paper/textbind-multi-turn-interleaved-multimodal","slug":"textbind-multi-turn-interleaved-multimodal","title":"TextBind: Multi-turn Interleaved Multimodal Instruction-following in the Wild","date":"2023-09-14","arxiv_id":"2309.08637","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-interpretable-cross-modal","slug":"a-survey-on-interpretable-cross-modal","title":"A Survey on Interpretable Cross-modal Reasoning","date":"2023-09-05","arxiv_id":"2309.01955","repositories_listed":1,"syntology":null},{"url":"/paper/towards-addressing-the-misalignment-of-object","slug":"towards-addressing-the-misalignment-of-object","title":"Towards Addressing the Misalignment of Object Proposal Evaluation for Vision-Language Tasks via Semantic Grounding","date":"2023-09-01","arxiv_id":"2309.00215","repositories_listed":1,"syntology":null},{"url":"/paper/separate-and-locate-rethink-the-text-in-text","slug":"separate-and-locate-rethink-the-text-in-text","title":"Separate and Locate: Rethink the Text in Text-based Visual Question Answering","date":"2023-08-31","arxiv_id":"2308.16383","repositories_listed":1,"syntology":null},{"url":"/paper/unipt-universal-parallel-tuning-for-transfer","slug":"unipt-universal-parallel-tuning-for-transfer","title":"UniPT: Universal Parallel Tuning for Transfer Learning with Efficient Parameter and Memory","date":"2023-08-28","arxiv_id":"2308.14316","repositories_listed":1,"syntology":null},{"url":"/paper/towards-vision-language-mechanistic","slug":"towards-vision-language-mechanistic","title":"Towards Vision-Language Mechanistic Interpretability: A Causal Tracing Tool for BLIP","date":"2023-08-27","arxiv_id":"2308.14179","repositories_listed":1,"syntology":null},{"url":"/paper/vqa-therapy-exploring-answer-differences-by","slug":"vqa-therapy-exploring-answer-differences-by","title":"VQA Therapy: Exploring Answer Differences by Visually Grounding Answers","date":"2023-08-21","arxiv_id":"2308.11662","repositories_listed":1,"syntology":null},{"url":"/paper/stablellava-enhanced-visual-instruction","slug":"stablellava-enhanced-visual-instruction","title":"StableLLaVA: Enhanced Visual Instruction Tuning with Synthesized Image-Dialogue Data","date":"2023-08-20","arxiv_id":"2308.10253","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stablellava-enhanced-visual-instruction#ran","syntology_url":"https://syntology.ai/paper/2308.10253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10253"}},"official":{"repos":["icoz69/stablellava"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bliva-a-simple-multimodal-llm-for-better","slug":"bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","arxiv_id":"2308.09936","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bliva-a-simple-multimodal-llm-for-better#ran","syntology_url":"https://syntology.ai/paper/2308.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09936"}},"official":{"repos":["mlpc-ucsd/bliva"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-the-meanings-of-function-words-from","slug":"learning-the-meanings-of-function-words-from","title":"Learning the meanings of function words from grounded language using a visual question answering model","date":"2023-08-16","arxiv_id":"2308.08628","repositories_listed":1,"syntology":null},{"url":"/paper/tech-text-guided-reconstruction-of-lifelike","slug":"tech-text-guided-reconstruction-of-lifelike","title":"TeCH: Text-guided Reconstruction of Lifelike Clothed Humans","date":"2023-08-16","arxiv_id":"2308.08545","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tech-text-guided-reconstruction-of-lifelike#ran","syntology_url":"https://syntology.ai/paper/2308.08545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08545"}},"official":{"repos":["huangyangyi/tech"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/detecting-and-preventing-hallucinations-in","slug":"detecting-and-preventing-hallucinations-in","title":"Detecting and Preventing Hallucinations in Large Vision Language Models","date":"2023-08-11","arxiv_id":"2308.06394","repositories_listed":1,"syntology":null},{"url":"/paper/foundation-model-is-efficient-multimodal-1","slug":"foundation-model-is-efficient-multimodal-1","title":"Foundation Model is Efficient Multimodal Multitask Model Selector","date":"2023-08-11","arxiv_id":"2308.06262","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/foundation-model-is-efficient-multimodal-1#ran","syntology_url":"https://syntology.ai/paper/2308.06262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06262"}},"official":{"repos":["opengvlab/multitask-model-selector"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/progressive-spatio-temporal-perception-for","slug":"progressive-spatio-temporal-perception-for","title":"Progressive Spatio-temporal Perception for Audio-Visual Question Answering","date":"2023-08-10","arxiv_id":"2308.05421","repositories_listed":1,"syntology":null},{"url":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn","slug":"scigraphqa-a-large-scale-synthetic-multi-turn","title":"SciGraphQA: A Large-Scale Synthetic Multi-Turn Question-Answering Dataset for Scientific Graphs","date":"2023-08-07","arxiv_id":"2308.03349","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn#ran","syntology_url":"https://syntology.ai/paper/2308.03349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03349"}},"official":{"repos":["findalexli/SciGraphQA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-generalist-foundation-model-for","slug":"towards-generalist-foundation-model-for","title":"Towards Generalist Foundation Model for Radiology by Leveraging Web-scale 2D&3D Medical Data","date":"2023-08-04","arxiv_id":"2308.02463","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-generalist-foundation-model-for#ran","syntology_url":"https://syntology.ai/paper/2308.02463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02463"}},"official":{"repos":["chaoyi-wu/radfm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/realcqa-scientific-chart-question-answering","slug":"realcqa-scientific-chart-question-answering","title":"RealCQA: Scientific Chart Question Answering as a Test-bed for First-Order Logic","date":"2023-08-03","arxiv_id":"2308.01979","repositories_listed":1,"syntology":null},{"url":"/paper/context-vqa-towards-context-aware-and","slug":"context-vqa-towards-context-aware-and","title":"Context-VQA: Towards Context-Aware and Purposeful Visual Question Answering","date":"2023-07-28","arxiv_id":"2307.15745","repositories_listed":1,"syntology":null},{"url":"/paper/rt-2-vision-language-action-models-transfer","slug":"rt-2-vision-language-action-models-transfer","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","date":"2023-07-28","arxiv_id":"2307.15818","repositories_listed":1,"syntology":null},{"url":"/paper/expert-knowledge-aware-image-difference-graph","slug":"expert-knowledge-aware-image-difference-graph","title":"Expert Knowledge-Aware Image Difference Graph Representation Learning for Difference-Aware Medical Visual Question Answering","date":"2023-07-22","arxiv_id":"2307.11986","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expert-knowledge-aware-image-difference-graph#ran","syntology_url":"https://syntology.ai/paper/2307.11986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11986"}},"official":{"repos":["holipori/mimic-diff-vqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/explaining-autonomous-driving-actions-with","slug":"explaining-autonomous-driving-actions-with","title":"Explaining Autonomous Driving Actions with Visual Question Answering","date":"2023-07-19","arxiv_id":"2307.10408","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-performance-analysis-on-pre-trained","slug":"towards-a-performance-analysis-on-pre-trained","title":"Towards a performance analysis on pre-trained Visual Question Answering models for autonomous driving","date":"2023-07-18","arxiv_id":"2307.09329","repositories_listed":1,"syntology":null},{"url":"/paper/co-attention-gated-vision-language-embedding","slug":"co-attention-gated-vision-language-embedding","title":"CAT-ViL: Co-Attention Gated Vision-Language Embedding for Visual Question Localized-Answering in Robotic Surgery","date":"2023-07-11","arxiv_id":"2307.05182","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-attention-gated-vision-language-embedding#ran","syntology_url":"https://syntology.ai/paper/2307.05182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05182"}},"official":{"repos":["longbai1006/cat-vil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rad-restruct-a-novel-vqa-benchmark-and-method","slug":"rad-restruct-a-novel-vqa-benchmark-and-method","title":"Rad-ReStruct: A Novel VQA Benchmark and Method for Structured Radiology Reporting","date":"2023-07-11","arxiv_id":"2307.05766","repositories_listed":1,"syntology":null},{"url":"/paper/journeydb-a-benchmark-for-generative-image","slug":"journeydb-a-benchmark-for-generative-image","title":"JourneyDB: A Benchmark for Generative Image Understanding","date":"2023-07-03","arxiv_id":"2307.00716","repositories_listed":1,"syntology":null},{"url":"/paper/localized-questions-in-medical-visual","slug":"localized-questions-in-medical-visual","title":"Localized Questions in Medical Visual Question Answering","date":"2023-07-03","arxiv_id":"2307.01067","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-prompt-retrieval-for-generative","slug":"multimodal-prompt-retrieval-for-generative","title":"Multimodal Prompt Retrieval for Generative Visual Question Answering","date":"2023-06-30","arxiv_id":"2306.17675","repositories_listed":1,"syntology":null},{"url":"/paper/answer-mining-from-a-pool-of-images-towards","slug":"answer-mining-from-a-pool-of-images-towards","title":"Answer Mining from a Pool of Images: Towards Retrieval-Based Visual Question Answering","date":"2023-06-29","arxiv_id":"2306.16713","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/answer-mining-from-a-pool-of-images-towards#ran","syntology_url":"https://syntology.ai/paper/2306.16713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16713"}},"official":{"repos":["Abhiram4572/mi_bart"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pre-training-multi-modal-dense-retrievers-for","slug":"pre-training-multi-modal-dense-retrievers-for","title":"Pre-Training Multi-Modal Dense Retrievers for Outside-Knowledge Visual Question Answering","date":"2023-06-28","arxiv_id":"2306.16478","repositories_listed":1,"syntology":null},{"url":"/paper/shikra-unleashing-multimodal-llm-s","slug":"shikra-unleashing-multimodal-llm-s","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","date":"2023-06-27","arxiv_id":"2306.15195","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shikra-unleashing-multimodal-llm-s#ran","syntology_url":"https://syntology.ai/paper/2306.15195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15195"}},"official":{"repos":["shikras/shikra"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-prompting-techniques-for-zero","slug":"investigating-prompting-techniques-for-zero","title":"Investigating Prompting Techniques for Zero- and Few-Shot Visual Question Answering","date":"2023-06-16","arxiv_id":"2306.09996","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/investigating-prompting-techniques-for-zero#ran","syntology_url":"https://syntology.ai/paper/2306.09996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09996"}},"official":{"repos":["rabiulcste/vqazero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/encyclopedic-vqa-visual-questions-about","slug":"encyclopedic-vqa-visual-questions-about","title":"Encyclopedic VQA: Visual questions about detailed properties of fine-grained categories","date":"2023-06-15","arxiv_id":"2306.09224","repositories_listed":1,"syntology":null},{"url":"/paper/lvlm-ehub-a-comprehensive-evaluation","slug":"lvlm-ehub-a-comprehensive-evaluation","title":"LVLM-eHub: A Comprehensive Evaluation Benchmark for Large Vision-Language Models","date":"2023-06-15","arxiv_id":"2306.09265","repositories_listed":1,"syntology":null},{"url":"/paper/improving-selective-visual-question-answering-1","slug":"improving-selective-visual-question-answering-1","title":"Improving Selective Visual Question Answering by Learning from Your Peers","date":"2023-06-14","arxiv_id":"2306.08751","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-neural-probabilistic-answer-set","slug":"scalable-neural-probabilistic-answer-set","title":"Scalable Neural-Probabilistic Answer Set Programming","date":"2023-06-14","arxiv_id":"2306.08397","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-neural-probabilistic-answer-set#ran","syntology_url":"https://syntology.ai/paper/2306.08397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08397"}},"official":{"repos":["ml-research/slash"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/safeguarding-data-in-multimodal-ai-a","slug":"safeguarding-data-in-multimodal-ai-a","title":"Safeguarding Data in Multimodal AI: A Differentially Private Approach to CLIP Training","date":"2023-06-13","arxiv_id":"2306.08173","repositories_listed":1,"syntology":null},{"url":"/paper/global-and-local-semantic-completion-learning","slug":"global-and-local-semantic-completion-learning","title":"Global and Local Semantic Completion Learning for Vision-Language Pre-training","date":"2023-06-12","arxiv_id":"2306.07096","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-pre-training-for-medical-vision","slug":"multi-modal-pre-training-for-medical-vision","title":"Multi-modal Pre-training for Medical Vision-language Understanding and Generation: An Empirical Study with A New Benchmark","date":"2023-06-10","arxiv_id":"2306.06494","repositories_listed":1,"syntology":null},{"url":"/paper/modular-visual-question-answering-via-code","slug":"modular-visual-question-answering-via-code","title":"Modular Visual Question Answering via Code Generation","date":"2023-06-08","arxiv_id":"2306.05392","repositories_listed":1,"syntology":null},{"url":"/paper/an-approach-to-solving-the-abstraction-and","slug":"an-approach-to-solving-the-abstraction-and","title":"An Approach to Solving the Abstraction and Reasoning Corpus (ARC) Challenge","date":"2023-06-06","arxiv_id":"2306.03553","repositories_listed":1,"syntology":null},{"url":"/paper/q-how-to-specialize-large-vision-language-1","slug":"q-how-to-specialize-large-vision-language-1","title":"Q: How to Specialize Large Vision-Language Models to Data-Scarce VQA Tasks? A: Self-Train on Unlabeled Images!","date":"2023-06-06","arxiv_id":"2306.03932","repositories_listed":1,"syntology":null},{"url":"/paper/visualgptscore-visio-linguistic-reasoning","slug":"visualgptscore-visio-linguistic-reasoning","title":"Revisiting the Role of Language Priors in Vision-Language Models","date":"2023-06-02","arxiv_id":"2306.01879","repositories_listed":1,"syntology":null},{"url":"/paper/llava-med-training-a-large-language-and","slug":"llava-med-training-a-large-language-and","title":"LLaVA-Med: Training a Large Language-and-Vision Assistant for Biomedicine in One Day","date":"2023-06-01","arxiv_id":"2306.00890","repositories_listed":1,"syntology":null},{"url":"/paper/dense-and-aligned-captions-dac-promote","slug":"dense-and-aligned-captions-dac-promote","title":"Dense and Aligned Captions (DAC) Promote Compositional Reasoning in VL Models","date":"2023-05-31","arxiv_id":"2305.19595","repositories_listed":1,"syntology":null},{"url":"/paper/multi-scale-attention-for-audio-question","slug":"multi-scale-attention-for-audio-question","title":"Multi-Scale Attention for Audio Question Answering","date":"2023-05-29","arxiv_id":"2305.17993","repositories_listed":1,"syntology":null},{"url":"/paper/havqa-a-dataset-for-visual-question-answering","slug":"havqa-a-dataset-for-visual-question-answering","title":"HaVQA: A Dataset for Visual Question Answering and Multimodal Research in Hausa Language","date":"2023-05-28","arxiv_id":"2305.17690","repositories_listed":1,"syntology":null},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/modularized-zero-shot-vqa-with-pre-trained","slug":"modularized-zero-shot-vqa-with-pre-trained","title":"Modularized Zero-shot VQA with Pre-trained Models","date":"2023-05-27","arxiv_id":"2305.17369","repositories_listed":1,"syntology":null},{"url":"/paper/biomedgpt-a-unified-and-generalist-biomedical","slug":"biomedgpt-a-unified-and-generalist-biomedical","title":"BiomedGPT: A Generalist Vision-Language Foundation Model for Diverse Biomedical Tasks","date":"2023-05-26","arxiv_id":"2305.17100","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/biomedgpt-a-unified-and-generalist-biomedical#ran","syntology_url":"https://syntology.ai/paper/2305.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17100"}},"official":{"repos":["taokz/biomedgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-visual-question-answering-with","slug":"zero-shot-visual-question-answering-with","title":"Zero-shot Visual Question Answering with Language Model Feedback","date":"2023-05-26","arxiv_id":"2305.17006","repositories_listed":1,"syntology":null},{"url":"/paper/cream-visually-situated-natural-language","slug":"cream-visually-situated-natural-language","title":"Visually-Situated Natural Language Understanding with Contrastive Reading Model and Frozen Large Language Models","date":"2023-05-24","arxiv_id":"2305.15080","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-faithful-and-plausible-visual","slug":"measuring-faithful-and-plausible-visual","title":"Measuring Faithful and Plausible Visual Grounding in VQA","date":"2023-05-24","arxiv_id":"2305.15015","repositories_listed":1,"syntology":null},{"url":"/paper/nuscenes-qa-a-multi-modal-visual-question","slug":"nuscenes-qa-a-multi-modal-visual-question","title":"NuScenes-QA: A Multi-modal Visual Question Answering Benchmark for Autonomous Driving Scenario","date":"2023-05-24","arxiv_id":"2305.14836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nuscenes-qa-a-multi-modal-visual-question#ran","syntology_url":"https://syntology.ai/paper/2305.14836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14836"}},"official":{"repos":["qiantianwen/nuscenes-qa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-art-of-socratic-questioning-zero-shot","slug":"the-art-of-socratic-questioning-zero-shot","title":"The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models","date":"2023-05-24","arxiv_id":"2305.14999","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/the-art-of-socratic-questioning-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2305.14999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14999"}},"official":{"repos":["vt-nlp/socratic-questioning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/memecap-a-dataset-for-captioning-and","slug":"memecap-a-dataset-for-captioning-and","title":"MemeCap: A Dataset for Captioning and Interpreting Memes","date":"2023-05-23","arxiv_id":"2305.13703","repositories_listed":1,"syntology":null},{"url":"/paper/target-aware-spatio-temporal-reasoning-via","slug":"target-aware-spatio-temporal-reasoning-via","title":"Target-Aware Spatio-Temporal Reasoning via Answering Questions in Dynamics Audio-Visual Scenarios","date":"2023-05-21","arxiv_id":"2305.12397","repositories_listed":1,"syntology":null}],"record_sha256":"9db23448ded68f783110c300c4dcab36b815cd05e6d2b07c1b753e4f3c48ae83","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}