{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/dataset/mm-vet/papers/2","list_of":"/dataset/mm-vet","dataset":"MM-Vet","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","key_notes":{"samples_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","samples_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"order":"archive","order_definition":"date (newest first), then slug","population":"every paper with a leaderboard row on this dataset's benchmarks (the benchmark-backed subset): the archive's own papers-using-this-dataset list was never published, so this is not that list; num_papers_in_archive is the archive's own count","page":2,"pages_in_order":2,"rows_per_page":100,"rows":[101,147],"of":147,"counts":{"papers_with_a_benchmark_row":147,"with_a_code_link":108,"where_syntology_ran_a_sample":69,"not_listed_spam_title":0,"listed":147,"listed_where_code_ran":69,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":57,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":57,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers with at least one leaderboard row on this dataset's benchmarks; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/dataset/mm-vet","prev":"/dataset/mm-vet","next":null,"papers":[{"paper":"/paper/tinyllava-a-framework-of-small-scale-large","slug":"tinyllava-a-framework-of-small-scale-large","title":"TinyLLaVA: A Framework of Small-scale Large Multimodal Models","date":"2024-02-22","arxiv_id":"2402.14289","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":4,"samples_unverified":4,"pointer_only_for_licence":1,"official":{"repos":["dlcv-buaa/tinyllavabench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/tinyllava-a-framework-of-small-scale-large#ran","syntology_url":"https://syntology.ai/paper/2402.14289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14289"}}}},{"paper":"/paper/collavo-crayon-large-language-and-vision","slug":"collavo-crayon-large-language-and-vision","title":"CoLLaVO: Crayon Large Language and Vision mOdel","date":"2024-02-17","arxiv_id":"2402.11248","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["ByungKwanLee/CoLLaVO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/collavo-crayon-large-language-and-vision#ran","syntology_url":"https://syntology.ai/paper/2402.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11248"}}}},{"paper":"/paper/sphinx-x-scaling-data-and-parameters-for-a","slug":"sphinx-x-scaling-data-and-parameters-for-a","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","date":"2024-02-08","arxiv_id":"2402.05935","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":3,"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/sphinx-x-scaling-data-and-parameters-for-a#ran","syntology_url":"https://syntology.ai/paper/2402.05935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05935"}}}},{"paper":"/paper/video-lavit-unified-video-language-pre","slug":"video-lavit-unified-video-language-pre","title":"Video-LaVIT: Unified Video-Language Pre-training with Decoupled Visual-Motional Tokenization","date":"2024-02-05","arxiv_id":"2402.03161","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":2,"pointer_only_for_licence":5,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/video-lavit-unified-video-language-pre#ran","syntology_url":"https://syntology.ai/paper/2402.03161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03161"}}}},{"paper":"/paper/enhancing-multimodal-large-language-models","slug":"enhancing-multimodal-large-language-models","title":"From Training-Free to Adaptive: Empirical Insights into MLLMs' Understanding of Detection Information","date":"2024-01-31","arxiv_id":"2401.17981","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/mousi-poly-visual-expert-vision-language","slug":"mousi-poly-visual-expert-vision-language","title":"MouSi: Poly-Visual-Expert Vision-Language Models","date":"2024-01-30","arxiv_id":"2401.17221","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/internlm-xcomposer2-mastering-free-form-text","slug":"internlm-xcomposer2-mastering-free-form-text","title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","date":"2024-01-29","arxiv_id":"2401.16420","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/moe-llava-mixture-of-experts-for-large-vision","slug":"moe-llava-mixture-of-experts-for-large-vision","title":"MoE-LLaVA: Mixture of Experts for Large Vision-Language Models","date":"2024-01-29","arxiv_id":"2401.15947","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":2,"official":{"repos":["PKU-YuanGroup/MoE-LLaVA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/moe-llava-mixture-of-experts-for-large-vision#ran","syntology_url":"https://syntology.ai/paper/2401.15947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15947"}}}},{"paper":"/paper/small-language-model-meets-with-reinforced","slug":"small-language-model-meets-with-reinforced","title":"Small Language Model Meets with Reinforced Vision Vocabulary","date":"2024-01-23","arxiv_id":"2401.12503","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/coco-is-all-you-need-for-visual-instruction","slug":"coco-is-all-you-need-for-visual-instruction","title":"COCO is \"ALL'' You Need for Visual Instruction Fine-tuning","date":"2024-01-17","arxiv_id":"2401.08968","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/camml-context-aware-multimodal-learner-for","slug":"camml-context-aware-multimodal-learner-for","title":"CaMML: Context-Aware Multimodal Learner for Large Models","date":"2024-01-06","arxiv_id":"2401.03149","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/llava-ph-efficient-multi-modal-assistant-with","slug":"llava-ph-efficient-multi-modal-assistant-with","title":"LLaVA-Phi: Efficient Multi-Modal Assistant with Small Language Model","date":"2024-01-04","arxiv_id":"2401.02330","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":3,"samples_unverified":3,"pointer_only_for_licence":9,"official":{"repos":["zhuyiche/llava-phi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/llava-ph-efficient-multi-modal-assistant-with#ran","syntology_url":"https://syntology.ai/paper/2401.02330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02330"}}}},{"paper":"/paper/textit-v-guided-visual-search-as-a-core","slug":"textit-v-guided-visual-search-as-a-core","title":"V*: Guided Visual Search as a Core Mechanism in Multimodal LLMs","date":"2023-12-21","arxiv_id":"2312.14135","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["penghao-wu/vstar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/textit-v-guided-visual-search-as-a-core#ran","syntology_url":"https://syntology.ai/paper/2312.14135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14135"}}}},{"paper":"/paper/generative-multimodal-models-are-in-context","slug":"generative-multimodal-models-are-in-context","title":"Generative Multimodal Models are In-Context Learners","date":"2023-12-20","arxiv_id":"2312.13286","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["baaivision/emu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/generative-multimodal-models-are-in-context#ran","syntology_url":"https://syntology.ai/paper/2312.13286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13286"}}}},{"paper":"/paper/gemini-a-family-of-highly-capable-multimodal-1","slug":"gemini-a-family-of-highly-capable-multimodal-1","title":"Gemini: A Family of Highly Capable Multimodal Models","date":"2023-12-19","arxiv_id":"2312.11805","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/silkie-preference-distillation-for-large","slug":"silkie-preference-distillation-for-large","title":"Silkie: Preference Distillation for Large Visual Language Models","date":"2023-12-17","arxiv_id":"2312.10665","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/cogagent-a-visual-language-model-for-gui","slug":"cogagent-a-visual-language-model-for-gui","title":"CogAgent: A Visual Language Model for GUI Agents","date":"2023-12-14","arxiv_id":"2312.08914","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":18,"samples_ran":12,"samples_constructed":0,"samples_ran_checked":12,"samples_ran_instrument_failed":0,"samples_unverified":6,"pointer_only_for_licence":1,"official":{"repos":["THUDM/CogAgent","thudm/cogvlm"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":6,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cogagent-a-visual-language-model-for-gui#ran","syntology_url":"https://syntology.ai/paper/2312.08914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08914"}}}},{"paper":"/paper/hallucination-augmented-contrastive-learning","slug":"hallucination-augmented-contrastive-learning","title":"Hallucination Augmented Contrastive Learning for Multimodal Large Language Model","date":"2023-12-12","arxiv_id":"2312.06968","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":3,"pointer_only_for_licence":5,"official":{"repos":["x-plug/mplug-halowl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hallucination-augmented-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2312.06968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06968"}}}},{"paper":"/paper/vila-on-pre-training-for-visual-language","slug":"vila-on-pre-training-for-visual-language","title":"VILA: On Pre-training for Visual Language Models","date":"2023-12-12","arxiv_id":"2312.07533","rows_on_this_dataset":1,"code_links":3,"syntology":null},{"paper":"/paper/vary-scaling-up-the-vision-vocabulary-for","slug":"vary-scaling-up-the-vision-vocabulary-for","title":"Vary: Scaling up the Vision Vocabulary for Large Vision-Language Models","date":"2023-12-11","arxiv_id":"2312.06109","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/onellm-one-framework-to-align-all-modalities","slug":"onellm-one-framework-to-align-all-modalities","title":"OneLLM: One Framework to Align All Modalities with Language","date":"2023-12-06","arxiv_id":"2312.03700","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":4,"official":{"repos":["csuhan/onellm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/onellm-one-framework-to-align-all-modalities#ran","syntology_url":"https://syntology.ai/paper/2312.03700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03700"}}}},{"paper":"/paper/merlin-empowering-multimodal-llms-with","slug":"merlin-empowering-multimodal-llms-with","title":"Merlin:Empowering Multimodal LLMs with Foresight Minds","date":"2023-11-30","arxiv_id":"2312.00589","rows_on_this_dataset":1,"code_links":0,"syntology":null},{"paper":"/paper/sharegpt4v-improving-large-multi-modal-models","slug":"sharegpt4v-improving-large-multi-modal-models","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","date":"2023-11-21","arxiv_id":"2311.12793","rows_on_this_dataset":2,"code_links":1,"syntology":null},{"paper":"/paper/video-llava-learning-united-visual-1","slug":"video-llava-learning-united-visual-1","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","date":"2023-11-16","arxiv_id":"2311.10122","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":2,"official":{"repos":["PKU-YuanGroup/Video-LLaVA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/video-llava-learning-united-visual-1#ran","syntology_url":"https://syntology.ai/paper/2311.10122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10122"}}}},{"paper":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and","slug":"sphinx-the-joint-mixing-of-weights-tasks-and","title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","date":"2023-11-13","arxiv_id":"2311.07575","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":3,"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and#ran","syntology_url":"https://syntology.ai/paper/2311.07575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07575"}}}},{"paper":"/paper/to-see-is-to-believe-prompting-gpt-4v-for","slug":"to-see-is-to-believe-prompting-gpt-4v-for","title":"To See is to Believe: Prompting GPT-4V for Better Visual Instruction Tuning","date":"2023-11-13","arxiv_id":"2311.07574","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/volcano-mitigating-multimodal-hallucination","slug":"volcano-mitigating-multimodal-hallucination","title":"Volcano: Mitigating Multimodal Hallucination through Self-Feedback Guided Revision","date":"2023-11-13","arxiv_id":"2311.07362","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":4,"samples_unverified":1,"pointer_only_for_licence":10,"official":{"repos":["kaistai/volcano"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/volcano-mitigating-multimodal-hallucination#ran","syntology_url":"https://syntology.ai/paper/2311.07362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07362"}}}},{"paper":"/paper/infmllm-a-unified-framework-for-visual","slug":"infmllm-a-unified-framework-for-visual","title":"InfMLLM: A Unified Framework for Visual-Language Tasks","date":"2023-11-12","arxiv_id":"2311.06791","rows_on_this_dataset":1,"code_links":2,"syntology":null},{"paper":"/paper/llava-plus-learning-to-use-tools-for-creating","slug":"llava-plus-learning-to-use-tools-for-creating","title":"LLaVA-Plus: Learning to Use Tools for Creating Multimodal Agents","date":"2023-11-09","arxiv_id":"2311.05437","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["LLaVA-VL/LLaVA-Plus-Codebase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/llava-plus-learning-to-use-tools-for-creating#ran","syntology_url":"https://syntology.ai/paper/2311.05437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05437"}}}},{"paper":"/paper/mplug-owl2-revolutionizing-multi-modal-large","slug":"mplug-owl2-revolutionizing-multi-modal-large","title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","date":"2023-11-07","arxiv_id":"2311.04257","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":3,"official":{"repos":["x-plug/mplug-owl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mplug-owl2-revolutionizing-multi-modal-large#ran","syntology_url":"https://syntology.ai/paper/2311.04257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04257"}}}},{"paper":"/paper/otterhd-a-high-resolution-multi-modality","slug":"otterhd-a-high-resolution-multi-modality","title":"OtterHD: A High-Resolution Multi-modality Model","date":"2023-11-07","arxiv_id":"2311.04219","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":14,"samples_ran":13,"samples_constructed":0,"samples_ran_checked":12,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":3,"official":{"repos":["luodian/otter"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/otterhd-a-high-resolution-multi-modality#ran","syntology_url":"https://syntology.ai/paper/2311.04219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04219"}}}},{"paper":"/paper/cogvlm-visual-expert-for-pretrained-language","slug":"cogvlm-visual-expert-for-pretrained-language","title":"CogVLM: Visual Expert for Pretrained Language Models","date":"2023-11-06","arxiv_id":"2311.03079","rows_on_this_dataset":2,"code_links":4,"syntology":null},{"paper":"/paper/improved-baselines-with-visual-instruction","slug":"improved-baselines-with-visual-instruction","title":"Improved Baselines with Visual Instruction Tuning","date":"2023-10-05","arxiv_id":"2310.03744","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":3,"samples_unverified":3,"pointer_only_for_licence":8,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/improved-baselines-with-visual-instruction#ran","syntology_url":"https://syntology.ai/paper/2310.03744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03744"}}}},{"paper":"/paper/dreamllm-synergistic-multimodal-comprehension","slug":"dreamllm-synergistic-multimodal-comprehension","title":"DreamLLM: Synergistic Multimodal Comprehension and Creation","date":"2023-09-20","arxiv_id":"2309.11499","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["RunpeiDong/DreamLLM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dreamllm-synergistic-multimodal-comprehension#ran","syntology_url":"https://syntology.ai/paper/2309.11499","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11499"}}}},{"paper":"/paper/an-empirical-study-of-scaling-instruct-tuned","slug":"an-empirical-study-of-scaling-instruct-tuned","title":"An Empirical Study of Scaling Instruct-Tuned Large Multimodal Models","date":"2023-09-18","arxiv_id":"2309.09958","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["haotian-liu/LLaVA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/an-empirical-study-of-scaling-instruct-tuned#ran","syntology_url":"https://syntology.ai/paper/2309.09958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.09958"}}}},{"paper":"/paper/textbind-multi-turn-interleaved-multimodal","slug":"textbind-multi-turn-interleaved-multimodal","title":"TextBind: Multi-turn Interleaved Multimodal Instruction-following in the Wild","date":"2023-09-14","arxiv_id":"2309.08637","rows_on_this_dataset":1,"code_links":1,"syntology":null},{"paper":"/paper/qwen-vl-a-frontier-large-vision-language","slug":"qwen-vl-a-frontier-large-vision-language","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","date":"2023-08-24","arxiv_id":"2308.12966","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":2,"official":{"repos":["qwenlm/qwen-vl"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/qwen-vl-a-frontier-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2308.12966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12966"}}}},{"paper":"/paper/stablellava-enhanced-visual-instruction","slug":"stablellava-enhanced-visual-instruction","title":"StableLLaVA: Enhanced Visual Instruction Tuning with Synthesized Image-Dialogue Data","date":"2023-08-20","arxiv_id":"2308.10253","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["icoz69/stablellava"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/stablellava-enhanced-visual-instruction#ran","syntology_url":"https://syntology.ai/paper/2308.10253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10253"}}}},{"paper":"/paper/openflamingo-an-open-source-framework-for","slug":"openflamingo-an-open-source-framework-for","title":"OpenFlamingo: An Open-Source Framework for Training Large Autoregressive Vision-Language Models","date":"2023-08-02","arxiv_id":"2308.01390","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/generative-pretraining-in-multimodality","slug":"generative-pretraining-in-multimodality","title":"Emu: Generative Pretraining in Multimodality","date":"2023-07-11","arxiv_id":"2307.05222","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["baaivision/emu"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/generative-pretraining-in-multimodality#ran","syntology_url":"https://syntology.ai/paper/2307.05222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05222"}}}},{"paper":"/paper/aligning-large-multi-modal-model-with-robust","slug":"aligning-large-multi-modal-model-with-robust","title":"Mitigating Hallucination in Large Multi-Modal Models via Robust Instruction Tuning","date":"2023-06-26","arxiv_id":"2306.14565","rows_on_this_dataset":1,"code_links":4,"syntology":null},{"paper":"/paper/mimic-it-multi-modal-in-context-instruction","slug":"mimic-it-multi-modal-in-context-instruction","title":"MIMIC-IT: Multi-Modal In-Context Instruction Tuning","date":"2023-06-08","arxiv_id":"2306.05425","rows_on_this_dataset":2,"code_links":2,"syntology":null},{"paper":"/paper/llama-adapter-v2-parameter-efficient-visual","slug":"llama-adapter-v2-parameter-efficient-visual","title":"LLaMA-Adapter V2: Parameter-Efficient Visual Instruction Model","date":"2023-04-28","arxiv_id":"2304.15010","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":0,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["zrrskywalker/llama-adapter"],"state":"official: not harvested","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/llama-adapter-v2-parameter-efficient-visual#ran","syntology_url":"https://syntology.ai/paper/2304.15010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.15010"}}}},{"paper":"/paper/minigpt-4-enhancing-vision-language","slug":"minigpt-4-enhancing-vision-language","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","date":"2023-04-20","arxiv_id":"2304.10592","rows_on_this_dataset":2,"code_links":6,"syntology":null},{"paper":"/paper/mm-react-prompting-chatgpt-for-multimodal","slug":"mm-react-prompting-chatgpt-for-multimodal","title":"MM-REACT: Prompting ChatGPT for Multimodal Reasoning and Action","date":"2023-03-20","arxiv_id":"2303.11381","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["microsoft/MM-REACT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mm-react-prompting-chatgpt-for-multimodal#ran","syntology_url":"https://syntology.ai/paper/2303.11381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11381"}}}},{"paper":"/paper/gpt-4-technical-report-1","slug":"gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","rows_on_this_dataset":5,"code_links":11,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["openai/evals"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/gpt-4-technical-report-1#ran","syntology_url":"https://syntology.ai/paper/2303.08774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08774"}}}},{"paper":"/paper/blip-2-bootstrapping-language-image-pre","slug":"blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","rows_on_this_dataset":1,"code_links":17,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":4,"samples_constructed":4,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":1,"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/blip-2-bootstrapping-language-image-pre#ran","syntology_url":"https://syntology.ai/paper/2301.12597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12597"}}}}],"record_sha256":"2a397af25630ade5d39c2b61af05792e4a4282755078ed1e0e4e5b413cdc02b3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}