{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering-1/papers/ran/2","list_of":"/task/visual-question-answering-1","task":"Visual Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":2,"pages_in_order":4,"rows_per_page":100,"rows":[101,200],"of":378,"counts":{"archive_papers_tagged":2177,"with_a_code_link":1042,"where_syntology_ran_a_sample":378,"not_listed_spam_title":0,"listed":2177,"listed_where_code_ran":378,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":308,"every_run_a_failure_of_syntologys_instrument":70,"listed_with_a_run_with_no_instrument_failure":308,"listed_every_run_a_failure_of_syntologys_instrument":70,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering-1/papers/ran/1","prev":"/task/visual-question-answering-1/papers/ran/1","next":"/task/visual-question-answering-1/papers/ran/3","papers":[{"url":"/paper/mixture-of-subspaces-in-low-rank-adaptation","slug":"mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.11909","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mixture-of-subspaces-in-low-rank-adaptation#ran","syntology_url":"https://syntology.ai/paper/2406.11909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11909"}},"official":{"repos":["wutaiqiang/moslora"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/yo-llava-your-personalized-language-and","slug":"yo-llava-your-personalized-language-and","title":"Yo'LLaVA: Your Personalized Language and Vision Assistant","date":"2024-06-13","arxiv_id":"2406.09400","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/yo-llava-your-personalized-language-and#ran","syntology_url":"https://syntology.ai/paper/2406.09400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09400"}},"official":{"repos":["WisconsinAIVision/YoLLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rs-agent-automating-remote-sensing-tasks","slug":"rs-agent-automating-remote-sensing-tasks","title":"RS-Agent: Automating Remote Sensing Tasks through Intelligent Agent","date":"2024-06-11","arxiv_id":"2406.07089","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs-agent-automating-remote-sensing-tasks#ran","syntology_url":"https://syntology.ai/paper/2406.07089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07089"}},"official":{"repos":["intellisensing/rs-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vcr-visual-caption-restoration","slug":"vcr-visual-caption-restoration","title":"VCR: A Task for Pixel-Level Complex Reasoning in Vision Language Models via Restoring Occluded Text","date":"2024-06-10","arxiv_id":"2406.06462","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vcr-visual-caption-restoration#ran","syntology_url":"https://syntology.ai/paper/2406.06462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.06462"}},"official":{"repos":["tianyu-z/vcr"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/wings-learning-multimodal-llms-without-text","slug":"wings-learning-multimodal-llms-without-text","title":"Wings: Learning Multimodal LLMs without Text-only Forgetting","date":"2024-06-05","arxiv_id":"2406.03496","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wings-learning-multimodal-llms-without-text#ran","syntology_url":"https://syntology.ai/paper/2406.03496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.03496"}},"official":null}},{"url":"/paper/dragonfly-multi-resolution-zoom-supercharges","slug":"dragonfly-multi-resolution-zoom-supercharges","title":"Dragonfly: Multi-Resolution Zoom-In Encoding Enhances Vision-Language Models","date":"2024-06-03","arxiv_id":"2406.00977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dragonfly-multi-resolution-zoom-supercharges#ran","syntology_url":"https://syntology.ai/paper/2406.00977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00977"}},"official":{"repos":["togethercomputer/dragonfly"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-large-vision-language-models-with","slug":"enhancing-large-vision-language-models-with","title":"Enhancing Large Vision Language Models with Self-Training on Image Comprehension","date":"2024-05-30","arxiv_id":"2405.19716","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-large-vision-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2405.19716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19716"}},"official":{"repos":["yihedeng9/stic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/instruction-guided-visual-masking","slug":"instruction-guided-visual-masking","title":"Instruction-Guided Visual Masking","date":"2024-05-30","arxiv_id":"2405.19783","repositories_listed":1,"syntology":{"n":28,"n_ran":20,"n_constructed":7,"n_ran_checked":11,"n_instrument":9,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":2,"phrase":"20 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 9 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/instruction-guided-visual-masking#ran","syntology_url":"https://syntology.ai/paper/2405.19783","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19783"}},"official":{"repos":["2toinf/ivm"],"state":"official (archive's flag): 20 ran","n_ran":20,"n_constructed":7,"n_ran_no_instrument_failure":11,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-object-hallucination-via-data","slug":"mitigating-object-hallucination-via-data","title":"Data-augmented phrase-level alignment for mitigating object hallucination","date":"2024-05-28","arxiv_id":"2405.18654","repositories_listed":0,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mitigating-object-hallucination-via-data#ran","syntology_url":"https://syntology.ai/paper/2405.18654","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.18654"}},"official":null}},{"url":"/paper/rlaif-v-aligning-mllms-through-open-source-ai","slug":"rlaif-v-aligning-mllms-through-open-source-ai","title":"RLAIF-V: Open-Source AI Feedback Leads to Super GPT-4V Trustworthiness","date":"2024-05-27","arxiv_id":"2405.17220","repositories_listed":5,"syntology":{"n":22,"n_ran":17,"n_constructed":0,"n_ran_checked":10,"n_instrument":7,"n_unverified":5,"n_honours":0,"n_violates":2,"n_no_contract":8,"n_pointer_only":17,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 2 violated, 8 with no contract checked; 7 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/rlaif-v-aligning-mllms-through-open-source-ai#ran","syntology_url":"https://syntology.ai/paper/2405.17220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17220"}},"official":{"repos":["openbmb/omnilmm","rlhf-v/rlaif-v"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/meteor-mamba-based-traversal-of-rationale-for","slug":"meteor-mamba-based-traversal-of-rationale-for","title":"Meteor: Mamba-based Traversal of Rationale for Large Language and Vision Models","date":"2024-05-24","arxiv_id":"2405.15574","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/meteor-mamba-based-traversal-of-rationale-for#ran","syntology_url":"https://syntology.ai/paper/2405.15574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15574"}},"official":{"repos":["byungkwanlee/meteor"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/convllava-hierarchical-backbones-as-visual","slug":"convllava-hierarchical-backbones-as-visual","title":"ConvLLaVA: Hierarchical Backbones as Visual Encoder for Large Multimodal Models","date":"2024-05-24","arxiv_id":"2405.15738","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/convllava-hierarchical-backbones-as-visual#ran","syntology_url":"https://syntology.ai/paper/2405.15738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15738"}},"official":{"repos":["alibaba/conv-llava"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-visual-language-modality-alignment","slug":"enhancing-visual-language-modality-alignment","title":"Enhancing Visual-Language Modality Alignment in Large Vision Language Models via Self-Improvement","date":"2024-05-24","arxiv_id":"2405.15973","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-visual-language-modality-alignment#ran","syntology_url":"https://syntology.ai/paper/2405.15973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15973"}},"official":{"repos":["umd-huang-lab/sima"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-mixture-of-experts-an-auto-tuning","slug":"dynamic-mixture-of-experts-an-auto-tuning","title":"Dynamic Mixture of Experts: An Auto-Tuning Approach for Efficient Transformer Models","date":"2024-05-23","arxiv_id":"2405.14297","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dynamic-mixture-of-experts-an-auto-tuning#ran","syntology_url":"https://syntology.ai/paper/2405.14297","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14297"}},"official":{"repos":["lins-lab/dynmoe"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/calibrated-self-rewarding-vision-language","slug":"calibrated-self-rewarding-vision-language","title":"Calibrated Self-Rewarding Vision Language Models","date":"2024-05-23","arxiv_id":"2405.14622","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/calibrated-self-rewarding-vision-language#ran","syntology_url":"https://syntology.ai/paper/2405.14622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14622"}},"official":{"repos":["yiyangzhou/csr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lova3-learning-to-visual-question-answering","slug":"lova3-learning-to-visual-question-answering","title":"LOVA3: Learning to Visual Question Answering, Asking and Assessment","date":"2024-05-23","arxiv_id":"2405.14974","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":2,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lova3-learning-to-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2405.14974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14974"}},"official":{"repos":["showlab/lova3"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/mtvqa-benchmarking-multilingual-text-centric","slug":"mtvqa-benchmarking-multilingual-text-centric","title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","date":"2024-05-20","arxiv_id":"2405.11985","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mtvqa-benchmarking-multilingual-text-centric#ran","syntology_url":"https://syntology.ai/paper/2405.11985","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11985"}},"official":{"repos":["bytedance/MTVQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imp-highly-capable-large-multimodal-models","slug":"imp-highly-capable-large-multimodal-models","title":"Imp: Highly Capable Large Multimodal Models for Mobile Devices","date":"2024-05-20","arxiv_id":"2405.12107","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imp-highly-capable-large-multimodal-models#ran","syntology_url":"https://syntology.ai/paper/2405.12107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12107"}},"official":{"repos":["milvlg/imp"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uni-moe-scaling-unified-multimodal-llms-with","slug":"uni-moe-scaling-unified-multimodal-llms-with","title":"Uni-MoE: Scaling Unified Multimodal LLMs with Mixture of Experts","date":"2024-05-18","arxiv_id":"2405.11273","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/uni-moe-scaling-unified-multimodal-llms-with#ran","syntology_url":"https://syntology.ai/paper/2405.11273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11273"}},"official":{"repos":["hitsz-tmg/umoe-scaling-unified-multimodal-llms"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/unirag-universal-retrieval-augmentation-for","slug":"unirag-universal-retrieval-augmentation-for","title":"UniRAG: Universal Retrieval Augmentation for Large Vision Language Models","date":"2024-05-16","arxiv_id":"2405.10311","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unirag-universal-retrieval-augmentation-for#ran","syntology_url":"https://syntology.ai/paper/2405.10311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10311"}},"official":{"repos":["castorini/unirag"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","slug":"cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","arxiv_id":"2405.05949","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":6,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled#ran","syntology_url":"https://syntology.ai/paper/2405.05949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05949"}},"official":{"repos":["shi-labs/cumo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/omnidrive-a-holistic-llm-agent-framework-for","slug":"omnidrive-a-holistic-llm-agent-framework-for","title":"OmniDrive: A Holistic Vision-Language Dataset for Autonomous Driving with Counterfactual Reasoning","date":"2024-05-02","arxiv_id":"2405.01533","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/omnidrive-a-holistic-llm-agent-framework-for#ran","syntology_url":"https://syntology.ai/paper/2405.01533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01533"}},"official":{"repos":["nvlabs/omnidrive"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tablevqa-bench-a-visual-question-answering","slug":"tablevqa-bench-a-visual-question-answering","title":"TableVQA-Bench: A Visual Question Answering Benchmark on Multiple Table Domains","date":"2024-04-30","arxiv_id":"2404.19205","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tablevqa-bench-a-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2404.19205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.19205"}},"official":{"repos":["naver-ai/tablevqabench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/meddr-diagnosis-guided-bootstrapping-for","slug":"meddr-diagnosis-guided-bootstrapping-for","title":"GSCo: Towards Generalizable AI in Medicine via Generalist-Specialist Collaboration","date":"2024-04-23","arxiv_id":"2404.15127","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/meddr-diagnosis-guided-bootstrapping-for#ran","syntology_url":"https://syntology.ai/paper/2404.15127","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15127"}},"official":{"repos":["sunanhe/meddr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/boter-bootstrapping-knowledge-selection-and","slug":"boter-bootstrapping-knowledge-selection-and","title":"Self-Bootstrapped Visual-Language Model for Knowledge Selection and Question Answering","date":"2024-04-22","arxiv_id":"2404.13947","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/boter-bootstrapping-knowledge-selection-and#ran","syntology_url":"https://syntology.ai/paper/2404.13947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13947"}},"official":{"repos":["haodongze/self-ksel-qans"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lapa-latent-prompt-assist-model-for-medical","slug":"lapa-latent-prompt-assist-model-for-medical","title":"LaPA: Latent Prompt Assist Model For Medical Visual Question Answering","date":"2024-04-19","arxiv_id":"2404.13039","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/lapa-latent-prompt-assist-model-for-medical#ran","syntology_url":"https://syntology.ai/paper/2404.13039","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.13039"}},"official":{"repos":["garygutc/lapa_model"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/look-listen-and-answer-overcoming-biases-for","slug":"look-listen-and-answer-overcoming-biases-for","title":"Look, Listen, and Answer: Overcoming Biases for Audio-Visual Question Answering","date":"2024-04-18","arxiv_id":"2404.12020","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/look-listen-and-answer-overcoming-biases-for#ran","syntology_url":"https://syntology.ai/paper/2404.12020","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.12020"}},"official":{"repos":["reml-group/music-avqa-r"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/self-supervised-visual-preference-alignment","slug":"self-supervised-visual-preference-alignment","title":"Self-Supervised Visual Preference Alignment","date":"2024-04-16","arxiv_id":"2404.10501","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/self-supervised-visual-preference-alignment#ran","syntology_url":"https://syntology.ai/paper/2404.10501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10501"}},"official":{"repos":["Kevinz-code/SeVa"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-vision-and-language-spaces-with","slug":"bridging-vision-and-language-spaces-with","title":"Bridging Vision and Language Spaces with Assignment Prediction","date":"2024-04-15","arxiv_id":"2404.09632","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-vision-and-language-spaces-with#ran","syntology_url":"https://syntology.ai/paper/2404.09632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.09632"}},"official":{"repos":["park-jungin/vlap"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-visual-question-answering-through","slug":"enhancing-visual-question-answering-through","title":"Enhancing Visual Question Answering through Question-Driven Image Captions as Prompts","date":"2024-04-12","arxiv_id":"2404.08589","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/enhancing-visual-question-answering-through#ran","syntology_url":"https://syntology.ai/paper/2404.08589","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08589"}},"official":{"repos":["ovguyo/captions-in-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ferret-v2-an-improved-baseline-for-referring","slug":"ferret-v2-an-improved-baseline-for-referring","title":"Ferret-v2: An Improved Baseline for Referring and Grounding with Large Language Models","date":"2024-04-11","arxiv_id":"2404.07973","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ferret-v2-an-improved-baseline-for-referring#ran","syntology_url":"https://syntology.ai/paper/2404.07973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07973"}},"official":null}},{"url":"/paper/view-selection-for-3d-captioning-via","slug":"view-selection-for-3d-captioning-via","title":"View Selection for 3D Captioning via Diffusion Ranking","date":"2024-04-11","arxiv_id":"2404.07984","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/view-selection-for-3d-captioning-via#ran","syntology_url":"https://syntology.ai/paper/2404.07984","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07984"}},"official":null}},{"url":"/paper/soft-prompting-with-graph-of-thought-for","slug":"soft-prompting-with-graph-of-thought-for","title":"Soft-Prompting with Graph-of-Thought for Multi-modal Representation Learning","date":"2024-04-06","arxiv_id":"2404.04538","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/soft-prompting-with-graph-of-thought-for#ran","syntology_url":"https://syntology.ai/paper/2404.04538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.04538"}},"official":{"repos":["shishicode/agot"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-by-correction-efficient-tuning-task","slug":"learning-by-correction-efficient-tuning-task","title":"Learning by Correction: Efficient Tuning Task for Zero-Shot Generative Vision-Language Reasoning","date":"2024-04-01","arxiv_id":"2404.00909","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":1,"n_ran_checked":4,"n_instrument":4,"n_unverified":2,"n_honours":1,"n_violates":2,"n_no_contract":1,"n_pointer_only":0,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 2 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-by-correction-efficient-tuning-task#ran","syntology_url":"https://syntology.ai/paper/2404.00909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00909"}},"official":{"repos":["shtuplus/iccc_cvpr2024"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/causalchaos-dataset-for-comprehensive-causal","slug":"causalchaos-dataset-for-comprehensive-causal","title":"CausalChaos! Dataset for Comprehensive Causal Action Question Answering Over Longer Causal Chains Grounded in Dynamic Visual Scenes","date":"2024-04-01","arxiv_id":"2404.01299","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causalchaos-dataset-for-comprehensive-causal#ran","syntology_url":"https://syntology.ai/paper/2404.01299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.01299"}},"official":{"repos":["lunaproject22/causalchaos"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/m3d-advancing-3d-medical-image-analysis-with","slug":"m3d-advancing-3d-medical-image-analysis-with","title":"M3D: Advancing 3D Medical Image Analysis with Multi-Modal Large Language Models","date":"2024-03-31","arxiv_id":"2404.00578","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/m3d-advancing-3d-medical-image-analysis-with#ran","syntology_url":"https://syntology.ai/paper/2404.00578","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00578"}},"official":{"repos":["baai-dcai/m3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/h2rsvlm-towards-helpful-and-honest-remote","slug":"h2rsvlm-towards-helpful-and-honest-remote","title":"VHM: Versatile and Honest Vision Language Model for Remote Sensing Image Analysis","date":"2024-03-29","arxiv_id":"2403.20213","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/h2rsvlm-towards-helpful-and-honest-remote#ran","syntology_url":"https://syntology.ai/paper/2403.20213","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20213"}},"official":{"repos":["opendatalab/h2rsvlm","opendatalab/vhm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unsolvable-problem-detection-evaluating","slug":"unsolvable-problem-detection-evaluating","title":"Unsolvable Problem Detection: Evaluating Trustworthiness of Vision Language Models","date":"2024-03-29","arxiv_id":"2403.20331","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/unsolvable-problem-detection-evaluating#ran","syntology_url":"https://syntology.ai/paper/2403.20331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.20331"}},"official":{"repos":["atsumiyai/upd"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-embeddings-the-promise-of-visual-table","slug":"beyond-embeddings-the-promise-of-visual-table","title":"Beyond Embeddings: The Promise of Visual Table in Visual Reasoning","date":"2024-03-27","arxiv_id":"2403.18252","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 2 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-embeddings-the-promise-of-visual-table#ran","syntology_url":"https://syntology.ai/paper/2403.18252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18252"}},"official":{"repos":["lavi-lab/visual-table"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/quantifying-and-mitigating-unimodal-biases-in","slug":"quantifying-and-mitigating-unimodal-biases-in","title":"Quantifying and Mitigating Unimodal Biases in Multimodal Large Language Models: A Causal Perspective","date":"2024-03-27","arxiv_id":"2403.18346","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/quantifying-and-mitigating-unimodal-biases-in#ran","syntology_url":"https://syntology.ai/paper/2403.18346","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18346"}},"official":{"repos":["opencausalab/more"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mini-gemini-mining-the-potential-of-multi","slug":"mini-gemini-mining-the-potential-of-multi","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","date":"2024-03-27","arxiv_id":"2403.18814","repositories_listed":2,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mini-gemini-mining-the-potential-of-multi#ran","syntology_url":"https://syntology.ai/paper/2403.18814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18814"}},"official":{"repos":["dvlab-research/minigemini"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/illusionvqa-a-challenging-optical-illusion","slug":"illusionvqa-a-challenging-optical-illusion","title":"IllusionVQA: A Challenging Optical Illusion Dataset for Vision Language Models","date":"2024-03-23","arxiv_id":"2403.15952","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/illusionvqa-a-challenging-optical-illusion#ran","syntology_url":"https://syntology.ai/paper/2403.15952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15952"}},"official":{"repos":["csebuetnlp/illusionvqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-prumerge-adaptive-token-reduction-for","slug":"llava-prumerge-adaptive-token-reduction-for","title":"LLaVA-PruMerge: Adaptive Token Reduction for Efficient Large Multimodal Models","date":"2024-03-22","arxiv_id":"2403.15388","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-prumerge-adaptive-token-reduction-for#ran","syntology_url":"https://syntology.ai/paper/2403.15388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.15388"}},"official":null}},{"url":"/paper/language-repository-for-long-video","slug":"language-repository-for-long-video","title":"Language Repository for Long Video Understanding","date":"2024-03-21","arxiv_id":"2403.14622","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-repository-for-long-video#ran","syntology_url":"https://syntology.ai/paper/2403.14622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14622"}},"official":{"repos":["kkahatapitiya/langrepo"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-spot-interactive-reasoning-improves","slug":"chain-of-spot-interactive-reasoning-improves","title":"Chain-of-Spot: Interactive Reasoning Improves Large Vision-Language Models","date":"2024-03-19","arxiv_id":"2403.12966","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chain-of-spot-interactive-reasoning-improves#ran","syntology_url":"https://syntology.ai/paper/2403.12966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12966"}},"official":{"repos":["dongyh20/chain-of-spot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-icl-bench-the-devil-in-the-details-of","slug":"vl-icl-bench-the-devil-in-the-details-of","title":"VL-ICL Bench: The Devil in the Details of Multimodal In-Context Learning","date":"2024-03-19","arxiv_id":"2403.13164","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vl-icl-bench-the-devil-in-the-details-of#ran","syntology_url":"https://syntology.ai/paper/2403.13164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13164"}},"official":{"repos":["ys-zong/vl-icl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/sq-llava-self-questioning-for-large-vision","slug":"sq-llava-self-questioning-for-large-vision","title":"SQ-LLaVA: Self-Questioning for Large Vision-Language Assistant","date":"2024-03-17","arxiv_id":"2403.11299","repositories_listed":2,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sq-llava-self-questioning-for-large-vision#ran","syntology_url":"https://syntology.ai/paper/2403.11299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11299"}},"official":{"repos":["heliossun/sq-llava"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/moai-mixture-of-all-intelligence-for-large","slug":"moai-mixture-of-all-intelligence-for-large","title":"MoAI: Mixture of All Intelligence for Large Language and Vision Models","date":"2024-03-12","arxiv_id":"2403.07508","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/moai-mixture-of-all-intelligence-for-large#ran","syntology_url":"https://syntology.ai/paper/2403.07508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07508"}},"official":{"repos":["ByungKwanLee/MoAI"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-text-frozen-large-language-models-in","slug":"beyond-text-frozen-large-language-models-in","title":"Beyond Text: Frozen Large Language Models in Visual Signal Comprehension","date":"2024-03-12","arxiv_id":"2403.07874","repositories_listed":1,"syntology":{"n":26,"n_ran":17,"n_constructed":0,"n_ran_checked":6,"n_instrument":11,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":26,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 11 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/beyond-text-frozen-large-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2403.07874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07874"}},"official":{"repos":["zh460045050/v2l-tokenizer"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/deepseek-vl-towards-real-world-vision","slug":"deepseek-vl-towards-real-world-vision","title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","date":"2024-03-08","arxiv_id":"2403.05525","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":7,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepseek-vl-towards-real-world-vision#ran","syntology_url":"https://syntology.ai/paper/2403.05525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05525"}},"official":{"repos":["deepseek-ai/deepseek-vl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cat-enhancing-multimodal-large-language-model","slug":"cat-enhancing-multimodal-large-language-model","title":"CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2024-03-07","arxiv_id":"2403.04640","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/cat-enhancing-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2403.04640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.04640"}},"official":{"repos":["rikeilong/bay-cat"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/feast-your-eyes-mixture-of-resolution","slug":"feast-your-eyes-mixture-of-resolution","title":"Feast Your Eyes: Mixture-of-Resolution Adaptation for Multimodal Large Language Models","date":"2024-03-05","arxiv_id":"2403.03003","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/feast-your-eyes-mixture-of-resolution#ran","syntology_url":"https://syntology.ai/paper/2403.03003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03003"}},"official":{"repos":["luogen1996/llava-hr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-all-seeing-project-v2-towards-general","slug":"the-all-seeing-project-v2-towards-general","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","date":"2024-02-29","arxiv_id":"2402.19474","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":2,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"7 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/the-all-seeing-project-v2-towards-general#ran","syntology_url":"https://syntology.ai/paper/2402.19474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19474"}},"official":{"repos":["opengvlab/all-seeing"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-the-gap-between-2d-and-3d-visual","slug":"bridging-the-gap-between-2d-and-3d-visual","title":"Bridging the Gap between 2D and 3D Visual Question Answering: A Fusion Approach for 3D VQA","date":"2024-02-24","arxiv_id":"2402.15933","repositories_listed":1,"syntology":{"n":13,"n_ran":12,"n_constructed":0,"n_ran_checked":8,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-the-gap-between-2d-and-3d-visual#ran","syntology_url":"https://syntology.ai/paper/2402.15933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15933"}},"official":{"repos":["matthewdm0816/bridgeqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tinyllava-a-framework-of-small-scale-large","slug":"tinyllava-a-framework-of-small-scale-large","title":"TinyLLaVA: A Framework of Small-scale Large Multimodal Models","date":"2024-02-22","arxiv_id":"2402.14289","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/tinyllava-a-framework-of-small-scale-large#ran","syntology_url":"https://syntology.ai/paper/2402.14289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14289"}},"official":{"repos":["dlcv-buaa/tinyllavabench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-aware-evaluation-for-vision","slug":"uncertainty-aware-evaluation-for-vision","title":"Uncertainty-Aware Evaluation for Vision-Language Models","date":"2024-02-22","arxiv_id":"2402.14418","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/uncertainty-aware-evaluation-for-vision#ran","syntology_url":"https://syntology.ai/paper/2402.14418","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14418"}},"official":{"repos":["ensec-ai/vlm-uncertainty-bench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-hallucinations-of-multi-modal-large","slug":"visual-hallucinations-of-multi-modal-large","title":"Visual Hallucinations of Multi-modal Large Language Models","date":"2024-02-22","arxiv_id":"2402.14683","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-hallucinations-of-multi-modal-large#ran","syntology_url":"https://syntology.ai/paper/2402.14683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14683"}},"official":{"repos":["wenhuang2000/vhtest"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cognitive-visual-language-mapper-advancing","slug":"cognitive-visual-language-mapper-advancing","title":"Cognitive Visual-Language Mapper: Advancing Multimodal Comprehension with Enhanced Visual Knowledge Alignment","date":"2024-02-21","arxiv_id":"2402.13561","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cognitive-visual-language-mapper-advancing#ran","syntology_url":"https://syntology.ai/paper/2402.13561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13561"}},"official":{"repos":["hitsz-tmg/cognitive-visual-language-mapper"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/aligning-modalities-in-vision-large-language","slug":"aligning-modalities-in-vision-large-language","title":"Aligning Modalities in Vision Large Language Models via Preference Fine-tuning","date":"2024-02-18","arxiv_id":"2402.11411","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/aligning-modalities-in-vision-large-language#ran","syntology_url":"https://syntology.ai/paper/2402.11411","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11411"}},"official":{"repos":["yiyangzhou/povid"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/collavo-crayon-large-language-and-vision","slug":"collavo-crayon-large-language-and-vision","title":"CoLLaVO: Crayon Large Language and Vision mOdel","date":"2024-02-17","arxiv_id":"2402.11248","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/collavo-crayon-large-language-and-vision#ran","syntology_url":"https://syntology.ai/paper/2402.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11248"}},"official":{"repos":["ByungKwanLee/CoLLaVO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-modal-preference-alignment-remedies","slug":"multi-modal-preference-alignment-remedies","title":"Multi-modal Preference Alignment Remedies Degradation of Visual Instruction Tuning on Language Models","date":"2024-02-16","arxiv_id":"2402.10884","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-modal-preference-alignment-remedies#ran","syntology_url":"https://syntology.ai/paper/2402.10884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10884"}},"official":{"repos":["findalexli/mllm-dpo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/vqattack-transferable-adversarial-attacks-on","slug":"vqattack-transferable-adversarial-attacks-on","title":"VQAttack: Transferable Adversarial Attacks on Visual Question Answering via Pre-trained Models","date":"2024-02-16","arxiv_id":"2402.11083","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vqattack-transferable-adversarial-attacks-on#ran","syntology_url":"https://syntology.ai/paper/2402.11083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11083"}},"official":null}},{"url":"/paper/omnimedvqa-a-new-large-scale-comprehensive","slug":"omnimedvqa-a-new-large-scale-comprehensive","title":"OmniMedVQA: A New Large-Scale Comprehensive Evaluation Benchmark for Medical LVLM","date":"2024-02-14","arxiv_id":"2402.09181","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/omnimedvqa-a-new-large-scale-comprehensive#ran","syntology_url":"https://syntology.ai/paper/2402.09181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09181"}},"official":{"repos":["opengvlab/multi-modality-arena"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/prismatic-vlms-investigating-the-design-space","slug":"prismatic-vlms-investigating-the-design-space","title":"Prismatic VLMs: Investigating the Design Space of Visually-Conditioned Language Models","date":"2024-02-12","arxiv_id":"2402.07865","repositories_listed":3,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prismatic-vlms-investigating-the-design-space#ran","syntology_url":"https://syntology.ai/paper/2402.07865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07865"}},"official":{"repos":["tri-ml/prismatic-vlms","tri-ml/vlm-evaluation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/a-benchmark-for-multi-modal-foundation-models","slug":"a-benchmark-for-multi-modal-foundation-models","title":"Q-Bench+: A Benchmark for Multi-modal Foundation Models on Low-level Vision from Single Images to Pairs","date":"2024-02-11","arxiv_id":"2402.07116","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-benchmark-for-multi-modal-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2402.07116","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07116"}},"official":{"repos":["Q-Future/Q-Bench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/open-ended-vqa-benchmarking-of-vision","slug":"open-ended-vqa-benchmarking-of-vision","title":"Open-ended VQA benchmarking of Vision-Language models by exploiting Classification datasets and their semantic hierarchy","date":"2024-02-11","arxiv_id":"2402.07270","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-ended-vqa-benchmarking-of-vision#ran","syntology_url":"https://syntology.ai/paper/2402.07270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07270"}},"official":{"repos":["lmb-freiburg/ovqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-graphs-meet-multi-modal-learning-a","slug":"knowledge-graphs-meet-multi-modal-learning-a","title":"Knowledge Graphs Meet Multi-Modal Learning: A Comprehensive Survey","date":"2024-02-08","arxiv_id":"2402.05391","repositories_listed":6,"syntology":{"n":18,"n_ran":15,"n_constructed":0,"n_ran_checked":13,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":4,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/knowledge-graphs-meet-multi-modal-learning-a#ran","syntology_url":"https://syntology.ai/paper/2402.05391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05391"}},"official":{"repos":["zjukg/kg-mm-survey"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/sphinx-x-scaling-data-and-parameters-for-a","slug":"sphinx-x-scaling-data-and-parameters-for-a","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","date":"2024-02-08","arxiv_id":"2402.05935","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-x-scaling-data-and-parameters-for-a#ran","syntology_url":"https://syntology.ai/paper/2402.05935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05935"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/video-lavit-unified-video-language-pre","slug":"video-lavit-unified-video-language-pre","title":"Video-LaVIT: Unified Video-Language Pre-training with Decoupled Visual-Motional Tokenization","date":"2024-02-05","arxiv_id":"2402.03161","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-lavit-unified-video-language-pre#ran","syntology_url":"https://syntology.ai/paper/2402.03161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03161"}},"official":null}},{"url":"/paper/gerea-question-aware-prompt-captions-for","slug":"gerea-question-aware-prompt-captions-for","title":"GeReA: Question-Aware Prompt Captions for Knowledge-based Visual Question Answering","date":"2024-02-04","arxiv_id":"2402.02503","repositories_listed":1,"syntology":{"n":18,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":18,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/gerea-question-aware-prompt-captions-for#ran","syntology_url":"https://syntology.ai/paper/2402.02503","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02503"}},"official":{"repos":["upper9527/gerea"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/question-answer-cross-language-image-matching","slug":"question-answer-cross-language-image-matching","title":"Question-Answer Cross Language Image Matching for Weakly Supervised Semantic Segmentation","date":"2024-01-18","arxiv_id":"2401.09883","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/question-answer-cross-language-image-matching#ran","syntology_url":"https://syntology.ai/paper/2401.09883","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09883"}},"official":{"repos":["cvi-szu/qa-clims"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/uncovering-the-full-potential-of-visual","slug":"uncovering-the-full-potential-of-visual","title":"Uncovering the Full Potential of Visual Grounding Methods in VQA","date":"2024-01-15","arxiv_id":"2401.07803","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uncovering-the-full-potential-of-visual#ran","syntology_url":"https://syntology.ai/paper/2401.07803","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.07803"}},"official":{"repos":["dreichcsl/truevg"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-retrieval-for-knowledge-based","slug":"cross-modal-retrieval-for-knowledge-based","title":"Cross-modal Retrieval for Knowledge-based Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05736","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/cross-modal-retrieval-for-knowledge-based#ran","syntology_url":"https://syntology.ai/paper/2401.05736","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05736"}},"official":{"repos":["paullerner/viquae"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/hallucination-benchmark-in-medical-visual","slug":"hallucination-benchmark-in-medical-visual","title":"Hallucination Benchmark in Medical Visual Question Answering","date":"2024-01-11","arxiv_id":"2401.05827","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hallucination-benchmark-in-medical-visual#ran","syntology_url":"https://syntology.ai/paper/2401.05827","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.05827"}},"official":{"repos":["knowlab/halt-medvqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-ph-efficient-multi-modal-assistant-with","slug":"llava-ph-efficient-multi-modal-assistant-with","title":"LLaVA-Phi: Efficient Multi-Modal Assistant with Small Language Model","date":"2024-01-04","arxiv_id":"2401.02330","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/llava-ph-efficient-multi-modal-assistant-with#ran","syntology_url":"https://syntology.ai/paper/2401.02330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02330"}},"official":{"repos":["zhuyiche/llava-phi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if","slug":"gpt-4v-ision-is-a-generalist-web-agent-if","title":"GPT-4V(ision) is a Generalist Web Agent, if Grounded","date":"2024-01-03","arxiv_id":"2401.01614","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gpt-4v-ision-is-a-generalist-web-agent-if#ran","syntology_url":"https://syntology.ai/paper/2401.01614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.01614"}},"official":{"repos":["osu-nlp-group/seeact"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tinygpt-v-efficient-multimodal-large-language","slug":"tinygpt-v-efficient-multimodal-large-language","title":"TinyGPT-V: Efficient Multimodal Large Language Model via Small Backbones","date":"2023-12-28","arxiv_id":"2312.16862","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tinygpt-v-efficient-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.16862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16862"}},"official":{"repos":["dlyuangod/tinygpt-v"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lingoqa-video-question-answering-for","slug":"lingoqa-video-question-answering-for","title":"LingoQA: Visual Question Answering for Autonomous Driving","date":"2023-12-21","arxiv_id":"2312.14115","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lingoqa-video-question-answering-for#ran","syntology_url":"https://syntology.ai/paper/2312.14115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14115"}},"official":{"repos":["wayveai/lingoqa"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/textit-v-guided-visual-search-as-a-core","slug":"textit-v-guided-visual-search-as-a-core","title":"V*: Guided Visual Search as a Core Mechanism in Multimodal LLMs","date":"2023-12-21","arxiv_id":"2312.14135","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textit-v-guided-visual-search-as-a-core#ran","syntology_url":"https://syntology.ai/paper/2312.14135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14135"}},"official":{"repos":["penghao-wu/vstar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vcoder-versatile-vision-encoders-for","slug":"vcoder-versatile-vision-encoders-for","title":"VCoder: Versatile Vision Encoders for Multimodal Large Language Models","date":"2023-12-21","arxiv_id":"2312.14233","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vcoder-versatile-vision-encoders-for#ran","syntology_url":"https://syntology.ai/paper/2312.14233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14233"}},"official":{"repos":["shi-labs/vcoder"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/internvl-scaling-up-vision-foundation-models","slug":"internvl-scaling-up-vision-foundation-models","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","date":"2023-12-21","arxiv_id":"2312.14238","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/internvl-scaling-up-vision-foundation-models#ran","syntology_url":"https://syntology.ai/paper/2312.14238","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14238"}},"official":{"repos":["opengvlab/internvl"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/generative-multimodal-models-are-in-context","slug":"generative-multimodal-models-are-in-context","title":"Generative Multimodal Models are In-Context Learners","date":"2023-12-20","arxiv_id":"2312.13286","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generative-multimodal-models-are-in-context#ran","syntology_url":"https://syntology.ai/paper/2312.13286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13286"}},"official":{"repos":["baaivision/emu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/earthvqa-towards-queryable-earth-via","slug":"earthvqa-towards-queryable-earth-via","title":"EarthVQA: Towards Queryable Earth via Relational Reasoning-Based Remote Sensing Visual Question Answering","date":"2023-12-19","arxiv_id":"2312.12222","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/earthvqa-towards-queryable-earth-via#ran","syntology_url":"https://syntology.ai/paper/2312.12222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.12222"}},"official":{"repos":["Junjue-Wang/EarthVQA"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/icd-lm-configuring-vision-language-in-context","slug":"icd-lm-configuring-vision-language-in-context","title":"Lever LM: Configuring In-Context Sequence to Lever Large Vision Language Models","date":"2023-12-15","arxiv_id":"2312.10104","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/icd-lm-configuring-vision-language-in-context#ran","syntology_url":"https://syntology.ai/paper/2312.10104","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10104"}},"official":{"repos":["forjadeforest/icd-lm","forjadeforest/lever-lm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/wordscape-a-pipeline-to-extract-multilingual-1","slug":"wordscape-a-pipeline-to-extract-multilingual-1","title":"WordScape: a Pipeline to extract multilingual, visually rich Documents with Layout Annotations from Web Crawl Data","date":"2023-12-15","arxiv_id":"2312.10188","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wordscape-a-pipeline-to-extract-multilingual-1#ran","syntology_url":"https://syntology.ai/paper/2312.10188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10188"}},"official":{"repos":["DS3Lab/WordScape"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advancing-surgical-vqa-with-scene-graph","slug":"advancing-surgical-vqa-with-scene-graph","title":"Advancing Surgical VQA with Scene Graph Knowledge","date":"2023-12-15","arxiv_id":"2312.10251","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/advancing-surgical-vqa-with-scene-graph#ran","syntology_url":"https://syntology.ai/paper/2312.10251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10251"}},"official":{"repos":["camma-public/ssg-qa","camma-public/ssg-vqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cogagent-a-visual-language-model-for-gui","slug":"cogagent-a-visual-language-model-for-gui","title":"CogAgent: A Visual Language Model for GUI Agents","date":"2023-12-14","arxiv_id":"2312.08914","repositories_listed":3,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/cogagent-a-visual-language-model-for-gui#ran","syntology_url":"https://syntology.ai/paper/2312.08914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08914"}},"official":{"repos":["THUDM/CogAgent","thudm/cogvlm"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/vlap-efficient-video-language-alignment-via","slug":"vlap-efficient-video-language-alignment-via","title":"ViLA: Efficient Video-Language Alignment for Video Question Answering","date":"2023-12-13","arxiv_id":"2312.08367","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlap-efficient-video-language-alignment-via#ran","syntology_url":"https://syntology.ai/paper/2312.08367","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08367"}},"official":{"repos":["xijun-cs/vila"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hallucination-augmented-contrastive-learning","slug":"hallucination-augmented-contrastive-learning","title":"Hallucination Augmented Contrastive Learning for Multimodal Large Language Model","date":"2023-12-12","arxiv_id":"2312.06968","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hallucination-augmented-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2312.06968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06968"}},"official":{"repos":["x-plug/mplug-halowl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/image-content-generation-with-causal","slug":"image-content-generation-with-causal","title":"Image Content Generation with Causal Reasoning","date":"2023-12-12","arxiv_id":"2312.07132","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/image-content-generation-with-causal#ran","syntology_url":"https://syntology.ai/paper/2312.07132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07132"}},"official":{"repos":["ieit-agi/mix-shannon"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/genixer-empowering-multimodal-large-language","slug":"genixer-empowering-multimodal-large-language","title":"Genixer: Empowering Multimodal Large Language Models as a Powerful Data Generator","date":"2023-12-11","arxiv_id":"2312.06731","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/genixer-empowering-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.06731","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06731"}},"official":{"repos":["zhaohengyuan1/genixer"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/language-informed-visual-concept-learning","slug":"language-informed-visual-concept-learning","title":"Language-Informed Visual Concept Learning","date":"2023-12-06","arxiv_id":"2312.03587","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/language-informed-visual-concept-learning#ran","syntology_url":"https://syntology.ai/paper/2312.03587","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03587"}},"official":{"repos":["sharonal10/langint"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/onellm-one-framework-to-align-all-modalities","slug":"onellm-one-framework-to-align-all-modalities","title":"OneLLM: One Framework to Align All Modalities with Language","date":"2023-12-06","arxiv_id":"2312.03700","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/onellm-one-framework-to-align-all-modalities#ran","syntology_url":"https://syntology.ai/paper/2312.03700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03700"}},"official":{"repos":["csuhan/onellm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchlmm-benchmarking-cross-style-visual","slug":"benchlmm-benchmarking-cross-style-visual","title":"BenchLMM: Benchmarking Cross-style Visual Capability of Large Multimodal Models","date":"2023-12-05","arxiv_id":"2312.02896","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchlmm-benchmarking-cross-style-visual#ran","syntology_url":"https://syntology.ai/paper/2312.02896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02896"}},"official":{"repos":["aifeg/benchgpt","aifeg/benchlmm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-to-configure-good-in-context-sequence-for","slug":"how-to-configure-good-in-context-sequence-for","title":"How to Configure Good In-Context Sequence for Visual Question Answering","date":"2023-12-04","arxiv_id":"2312.01571","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-to-configure-good-in-context-sequence-for#ran","syntology_url":"https://syntology.ai/paper/2312.01571","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.01571"}},"official":{"repos":["garyjiajia/ofv2_icl_vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/recursive-visual-programming","slug":"recursive-visual-programming","title":"Recursive Visual Programming","date":"2023-12-04","arxiv_id":"2312.02249","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/recursive-visual-programming#ran","syntology_url":"https://syntology.ai/paper/2312.02249","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02249"}},"official":{"repos":["para-lost/rvp"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large","slug":"llama-vid-an-image-is-worth-2-tokens-in-large","title":"LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models","date":"2023-11-28","arxiv_id":"2311.17043","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llama-vid-an-image-is-worth-2-tokens-in-large#ran","syntology_url":"https://syntology.ai/paper/2311.17043","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17043"}},"official":{"repos":["dvlab-research/llama-vid"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/geochat-grounded-large-vision-language-model","slug":"geochat-grounded-large-vision-language-model","title":"GeoChat: Grounded Large Vision-Language Model for Remote Sensing","date":"2023-11-24","arxiv_id":"2311.15826","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/geochat-grounded-large-vision-language-model#ran","syntology_url":"https://syntology.ai/paper/2311.15826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15826"}},"official":{"repos":["mbzuai-oryx/geochat"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/video-llava-learning-united-visual-1","slug":"video-llava-learning-united-visual-1","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","date":"2023-11-16","arxiv_id":"2311.10122","repositories_listed":6,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-llava-learning-united-visual-1#ran","syntology_url":"https://syntology.ai/paper/2311.10122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10122"}},"official":{"repos":["PKU-YuanGroup/Video-LLaVA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/attribute-diversity-determines-the","slug":"attribute-diversity-determines-the","title":"Attribute Diversity Determines the Systematicity Gap in VQA","date":"2023-11-15","arxiv_id":"2311.08695","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attribute-diversity-determines-the#ran","syntology_url":"https://syntology.ai/paper/2311.08695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08695"}},"official":{"repos":["ikb-a/systematicity-gap-in-vqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"ccd2e7de5b2aeda153084a635e2b9f8e861b47ba0930171e1509e1f49d810a00","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}