{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/dataset/mm-vet/papers/ran/1","list_of":"/dataset/mm-vet","dataset":"MM-Vet","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","key_notes":{"samples_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","samples_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this dataset or check it against this dataset's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","population":"every paper with a leaderboard row on this dataset's benchmarks (the benchmark-backed subset): the archive's own papers-using-this-dataset list was never published, so this is not that list; num_papers_in_archive is the archive's own count","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,69],"of":69,"counts":{"papers_with_a_benchmark_row":147,"with_a_code_link":108,"where_syntology_ran_a_sample":69,"not_listed_spam_title":0,"listed":147,"listed_where_code_ran":69,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":57,"every_run_a_failure_of_syntologys_instrument":12,"listed_with_a_run_with_no_instrument_failure":57,"listed_every_run_a_failure_of_syntologys_instrument":12,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers with at least one leaderboard row on this dataset's benchmarks; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/dataset/mm-vet/papers/ran/1","prev":null,"next":null,"papers":[{"paper":"/paper/lyra-an-efficient-and-speech-centric","slug":"lyra-an-efficient-and-speech-centric","title":"Lyra: An Efficient and Speech-Centric Framework for Omni-Cognition","date":"2024-12-12","arxiv_id":"2412.09501","rows_on_this_dataset":4,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":19,"samples_ran":14,"samples_constructed":0,"samples_ran_checked":11,"samples_ran_instrument_failed":3,"samples_unverified":5,"pointer_only_for_licence":4,"official":{"repos":["dvlab-research/Lyra"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/lyra-an-efficient-and-speech-centric#ran","syntology_url":"https://syntology.ai/paper/2412.09501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09501"}}}},{"paper":"/paper/provision-programmatically-scaling-vision","slug":"provision-programmatically-scaling-vision","title":"ProVision: Programmatically Scaling Vision-centric Instruction Data for Multimodal Language Models","date":"2024-12-09","arxiv_id":"2412.07012","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":15,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":0,"samples_unverified":8,"pointer_only_for_licence":0,"official":{"repos":["jieyuz2/provision"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":8,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/provision-programmatically-scaling-vision#ran","syntology_url":"https://syntology.ai/paper/2412.07012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.07012"}}}},{"paper":"/paper/linvt-empower-your-image-level-large-language","slug":"linvt-empower-your-image-level-large-language","title":"LinVT: Empower Your Image-level Large Language Model to Understand Videos","date":"2024-12-06","arxiv_id":"2412.05185","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":9,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":12,"official":{"repos":["gls0425/linvt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/linvt-empower-your-image-level-large-language#ran","syntology_url":"https://syntology.ai/paper/2412.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05185"}}}},{"paper":"/paper/expanding-performance-boundaries-of-open","slug":"expanding-performance-boundaries-of-open","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","date":"2024-12-06","arxiv_id":"2412.05271","rows_on_this_dataset":7,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":0,"samples_unverified":8,"pointer_only_for_licence":0,"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/expanding-performance-boundaries-of-open#ran","syntology_url":"https://syntology.ai/paper/2412.05271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.05271"}}}},{"paper":"/paper/flashsloth-lightning-multimodal-large","slug":"flashsloth-lightning-multimodal-large","title":"FlashSloth: Lightning Multimodal Large Language Models via Embedded Visual Compression","date":"2024-12-05","arxiv_id":"2412.04317","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":4,"samples_unverified":0,"pointer_only_for_licence":7,"official":{"repos":["codefanw/flashsloth"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/flashsloth-lightning-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2412.04317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.04317"}}}},{"paper":"/paper/visionzip-longer-is-better-but-not-necessary","slug":"visionzip-longer-is-better-but-not-necessary","title":"VisionZip: Longer is Better but Not Necessary in Vision Language Models","date":"2024-12-05","arxiv_id":"2412.04467","rows_on_this_dataset":6,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":1,"samples_unverified":3,"pointer_only_for_licence":0,"official":{"repos":["dvlab-research/visionzip"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/visionzip-longer-is-better-but-not-necessary#ran","syntology_url":"https://syntology.ai/paper/2412.04467","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.04467"}}}},{"paper":"/paper/a-stitch-in-time-saves-nine-small-vlm-is-a","slug":"a-stitch-in-time-saves-nine-small-vlm-is-a","title":"A Stitch in Time Saves Nine: Small VLM is a Precise Guidance for Accelerating Large VLMs","date":"2024-12-04","arxiv_id":"2412.03324","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":5,"official":{"repos":["NUS-HPC-AI-Lab/SGL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/a-stitch-in-time-saves-nine-small-vlm-is-a#ran","syntology_url":"https://syntology.ai/paper/2412.03324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.03324"}}}},{"paper":"/paper/dynamic-llava-efficient-multimodal-large","slug":"dynamic-llava-efficient-multimodal-large","title":"Dynamic-LLaVA: Efficient Multimodal Large Language Models via Dynamic Vision-language Context Sparsification","date":"2024-12-01","arxiv_id":"2412.00876","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":18,"samples_ran":13,"samples_constructed":4,"samples_ran_checked":5,"samples_ran_instrument_failed":8,"samples_unverified":5,"pointer_only_for_licence":0,"official":{"repos":["osilly/dynamic_llava"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dynamic-llava-efficient-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2412.00876","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.00876"}}}},{"paper":"/paper/vlfeedback-a-large-scale-ai-feedback-dataset","slug":"vlfeedback-a-large-scale-ai-feedback-dataset","title":"VLFeedback: A Large-Scale AI Feedback Dataset for Large Vision-Language Models Alignment","date":"2024-10-12","arxiv_id":"2410.09421","rows_on_this_dataset":4,"code_links":0,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":5,"samples_unverified":2,"pointer_only_for_licence":1,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/vlfeedback-a-large-scale-ai-feedback-dataset#ran","syntology_url":"https://syntology.ai/paper/2410.09421","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09421"}}}},{"paper":"/paper/deciphering-cross-modal-alignment-in-large","slug":"deciphering-cross-modal-alignment-in-large","title":"Deciphering Cross-Modal Alignment in Large Vision-Language Models with Modality Integration Rate","date":"2024-10-09","arxiv_id":"2410.07167","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":12,"samples_constructed":0,"samples_ran_checked":8,"samples_ran_instrument_failed":4,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["shikiw/modality-integration-rate"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/deciphering-cross-modal-alignment-in-large#ran","syntology_url":"https://syntology.ai/paper/2410.07167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07167"}}}},{"paper":"/paper/emu3-next-token-prediction-is-all-you-need","slug":"emu3-next-token-prediction-is-all-you-need","title":"Emu3: Next-Token Prediction is All You Need","date":"2024-09-27","arxiv_id":"2409.18869","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":0,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/emu3-next-token-prediction-is-all-you-need#ran","syntology_url":"https://syntology.ai/paper/2409.18869","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18869"}}}},{"paper":"/paper/phantom-of-latent-for-large-language-and","slug":"phantom-of-latent-for-large-language-and","title":"Phantom of Latent for Large Language and Vision Models","date":"2024-09-23","arxiv_id":"2409.14713","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["byungkwanlee/phantom"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/phantom-of-latent-for-large-language-and#ran","syntology_url":"https://syntology.ai/paper/2409.14713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.14713"}}}},{"paper":"/paper/qwen2-vl-enhancing-vision-language-model-s","slug":"qwen2-vl-enhancing-vision-language-model-s","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","date":"2024-09-18","arxiv_id":"2409.12191","rows_on_this_dataset":3,"code_links":8,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":12,"samples_constructed":1,"samples_ran_checked":10,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["qwenlm/qwen2-vl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/qwen2-vl-enhancing-vision-language-model-s#ran","syntology_url":"https://syntology.ai/paper/2409.12191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12191"}}}},{"paper":"/paper/cogvlm2-visual-language-models-for-image-and","slug":"cogvlm2-visual-language-models-for-image-and","title":"CogVLM2: Visual Language Models for Image and Video Understanding","date":"2024-08-29","arxiv_id":"2408.16500","rows_on_this_dataset":2,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":17,"samples_ran":12,"samples_constructed":0,"samples_ran_checked":12,"samples_ran_instrument_failed":0,"samples_unverified":5,"pointer_only_for_licence":0,"official":{"repos":["thudm/cogvlm2","thudm/glm-4"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":5,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cogvlm2-visual-language-models-for-image-and#ran","syntology_url":"https://syntology.ai/paper/2408.16500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.16500"}}}},{"paper":"/paper/visual-agents-as-fast-and-slow-thinkers","slug":"visual-agents-as-fast-and-slow-thinkers","title":"Visual Agents as Fast and Slow Thinkers","date":"2024-08-16","arxiv_id":"2408.08862","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["guangyans/sys2-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/visual-agents-as-fast-and-slow-thinkers#ran","syntology_url":"https://syntology.ai/paper/2408.08862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08862"}}}},{"paper":"/paper/inf-llava-dual-perspective-perception-for","slug":"inf-llava-dual-perspective-perception-for","title":"INF-LLaVA: Dual-perspective Perception for High-Resolution Multimodal Large Language Model","date":"2024-07-23","arxiv_id":"2407.16198","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["weihuanglin/inf-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/inf-llava-dual-perspective-perception-for#ran","syntology_url":"https://syntology.ai/paper/2407.16198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.16198"}}}},{"paper":"/paper/mminstruct-a-high-quality-multi-modal","slug":"mminstruct-a-high-quality-multi-modal","title":"MMInstruct: A High-Quality Multi-Modal Instruction Tuning Dataset with Extensive Diversity","date":"2024-07-22","arxiv_id":"2407.15838","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["yuecao0119/mminstruct"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mminstruct-a-high-quality-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2407.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15838"}}}},{"paper":"/paper/densefusion-1m-merging-vision-experts-for","slug":"densefusion-1m-merging-vision-experts-for","title":"DenseFusion-1M: Merging Vision Experts for Comprehensive Multimodal Perception","date":"2024-07-11","arxiv_id":"2407.08303","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":1,"pointer_only_for_licence":4,"official":{"repos":["baaivision/densefusion"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/densefusion-1m-merging-vision-experts-for#ran","syntology_url":"https://syntology.ai/paper/2407.08303","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.08303"}}}},{"paper":"/paper/internlm-xcomposer-2-5-a-versatile-large","slug":"internlm-xcomposer-2-5-a-versatile-large","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","date":"2024-07-03","arxiv_id":"2407.03320","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["internlm/internlm-xcomposer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/internlm-xcomposer-2-5-a-versatile-large#ran","syntology_url":"https://syntology.ai/paper/2407.03320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.03320"}}}},{"paper":"/paper/tokenpacker-efficient-visual-projector-for","slug":"tokenpacker-efficient-visual-projector-for","title":"TokenPacker: Efficient Visual Projector for Multimodal LLM","date":"2024-07-02","arxiv_id":"2407.02392","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["circleradon/tokenpacker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/tokenpacker-efficient-visual-projector-for#ran","syntology_url":"https://syntology.ai/paper/2407.02392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02392"}}}},{"paper":"/paper/llavolta-efficient-multi-modal-models-via","slug":"llavolta-efficient-multi-modal-models-via","title":"Efficient Large Multi-modal Models via Visual Context Compression","date":"2024-06-28","arxiv_id":"2406.20092","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":4,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["beckschen/llavolta"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/llavolta-efficient-multi-modal-models-via#ran","syntology_url":"https://syntology.ai/paper/2406.20092","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.20092"}}}},{"paper":"/paper/mg-llava-towards-multi-granularity-visual","slug":"mg-llava-towards-multi-granularity-visual","title":"MG-LLaVA: Towards Multi-Granularity Visual Instruction Tuning","date":"2024-06-25","arxiv_id":"2406.17770","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["phoenixz810/mg-llava"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mg-llava-towards-multi-granularity-visual#ran","syntology_url":"https://syntology.ai/paper/2406.17770","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17770"}}}},{"paper":"/paper/trol-traversal-of-layers-for-large-language","slug":"trol-traversal-of-layers-for-large-language","title":"TroL: Traversal of Layers for Large Language and Vision Models","date":"2024-06-18","arxiv_id":"2406.12246","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":30,"samples_ran":23,"samples_constructed":11,"samples_ran_checked":15,"samples_ran_instrument_failed":8,"samples_unverified":7,"pointer_only_for_licence":30,"official":{"repos":["byungkwanlee/trol"],"state":"official (archive's flag): 23 ran","n_ran":23,"n_constructed":11,"n_ran_no_instrument_failure":15,"n_unverified":7,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/trol-traversal-of-layers-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.12246","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12246"}}}},{"paper":"/paper/mmdu-a-multi-turn-multi-image-dialog","slug":"mmdu-a-multi-turn-multi-image-dialog","title":"MMDU: A Multi-Turn Multi-Image Dialog Understanding Benchmark and Instruction-Tuning Dataset for LVLMs","date":"2024-06-17","arxiv_id":"2406.11833","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":13,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":3,"samples_unverified":6,"pointer_only_for_licence":6,"official":{"repos":["liuziyu77/mmdu"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mmdu-a-multi-turn-multi-image-dialog#ran","syntology_url":"https://syntology.ai/paper/2406.11833","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11833"}}}},{"paper":"/paper/mixture-of-subspaces-in-low-rank-adaptation","slug":"mixture-of-subspaces-in-low-rank-adaptation","title":"Mixture-of-Subspaces in Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.11909","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":3,"samples_unverified":2,"pointer_only_for_licence":6,"official":{"repos":["wutaiqiang/moslora"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mixture-of-subspaces-in-low-rank-adaptation#ran","syntology_url":"https://syntology.ai/paper/2406.11909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11909"}}}},{"paper":"/paper/dragonfly-multi-resolution-zoom-supercharges","slug":"dragonfly-multi-resolution-zoom-supercharges","title":"Dragonfly: Multi-Resolution Zoom-In Encoding Enhances Vision-Language Models","date":"2024-06-03","arxiv_id":"2406.00977","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["togethercomputer/dragonfly"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dragonfly-multi-resolution-zoom-supercharges#ran","syntology_url":"https://syntology.ai/paper/2406.00977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00977"}}}},{"paper":"/paper/enhancing-large-vision-language-models-with","slug":"enhancing-large-vision-language-models-with","title":"Enhancing Large Vision Language Models with Self-Training on Image Comprehension","date":"2024-05-30","arxiv_id":"2405.19716","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["yihedeng9/stic"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/enhancing-large-vision-language-models-with#ran","syntology_url":"https://syntology.ai/paper/2405.19716","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19716"}}}},{"paper":"/paper/meteor-mamba-based-traversal-of-rationale-for","slug":"meteor-mamba-based-traversal-of-rationale-for","title":"Meteor: Mamba-based Traversal of Rationale for Large Language and Vision Models","date":"2024-05-24","arxiv_id":"2405.15574","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["byungkwanlee/meteor"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/meteor-mamba-based-traversal-of-rationale-for#ran","syntology_url":"https://syntology.ai/paper/2405.15574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15574"}}}},{"paper":"/paper/convllava-hierarchical-backbones-as-visual","slug":"convllava-hierarchical-backbones-as-visual","title":"ConvLLaVA: Hierarchical Backbones as Visual Encoder for Large Multimodal Models","date":"2024-05-24","arxiv_id":"2405.15738","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":3,"samples_unverified":2,"pointer_only_for_licence":1,"official":{"repos":["alibaba/conv-llava"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/convllava-hierarchical-backbones-as-visual#ran","syntology_url":"https://syntology.ai/paper/2405.15738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15738"}}}},{"paper":"/paper/enhancing-visual-language-modality-alignment","slug":"enhancing-visual-language-modality-alignment","title":"Enhancing Visual-Language Modality Alignment in Large Vision Language Models via Self-Improvement","date":"2024-05-24","arxiv_id":"2405.15973","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["umd-huang-lab/sima"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/enhancing-visual-language-modality-alignment#ran","syntology_url":"https://syntology.ai/paper/2405.15973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15973"}}}},{"paper":"/paper/dynamic-mixture-of-experts-an-auto-tuning","slug":"dynamic-mixture-of-experts-an-auto-tuning","title":"Dynamic Mixture of Experts: An Auto-Tuning Approach for Efficient Transformer Models","date":"2024-05-23","arxiv_id":"2405.14297","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["lins-lab/dynmoe"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dynamic-mixture-of-experts-an-auto-tuning#ran","syntology_url":"https://syntology.ai/paper/2405.14297","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14297"}}}},{"paper":"/paper/calibrated-self-rewarding-vision-language","slug":"calibrated-self-rewarding-vision-language","title":"Calibrated Self-Rewarding Vision Language Models","date":"2024-05-23","arxiv_id":"2405.14622","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":1,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["yiyangzhou/csr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/calibrated-self-rewarding-vision-language#ran","syntology_url":"https://syntology.ai/paper/2405.14622","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14622"}}}},{"paper":"/paper/lova3-learning-to-visual-question-answering","slug":"lova3-learning-to-visual-question-answering","title":"LOVA3: Learning to Visual Question Answering, Asking and Assessment","date":"2024-05-23","arxiv_id":"2405.14974","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":3,"samples_unverified":2,"pointer_only_for_licence":10,"official":{"repos":["showlab/lova3"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/lova3-learning-to-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2405.14974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14974"}}}},{"paper":"/paper/imp-highly-capable-large-multimodal-models","slug":"imp-highly-capable-large-multimodal-models","title":"Imp: Highly Capable Large Multimodal Models for Mobile Devices","date":"2024-05-20","arxiv_id":"2405.12107","rows_on_this_dataset":3,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["milvlg/imp"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/imp-highly-capable-large-multimodal-models#ran","syntology_url":"https://syntology.ai/paper/2405.12107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.12107"}}}},{"paper":"/paper/uni-moe-scaling-unified-multimodal-llms-with","slug":"uni-moe-scaling-unified-multimodal-llms-with","title":"Uni-MoE: Scaling Unified Multimodal LLMs with Mixture of Experts","date":"2024-05-18","arxiv_id":"2405.11273","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":3,"samples_unverified":2,"pointer_only_for_licence":8,"official":{"repos":["hitsz-tmg/umoe-scaling-unified-multimodal-llms"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/uni-moe-scaling-unified-multimodal-llms-with#ran","syntology_url":"https://syntology.ai/paper/2405.11273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11273"}}}},{"paper":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled","slug":"cumo-scaling-multimodal-llm-with-co-upcycled","title":"CuMo: Scaling Multimodal LLM with Co-Upcycled Mixture-of-Experts","date":"2024-05-09","arxiv_id":"2405.05949","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":12,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":5,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["shi-labs/cumo"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cumo-scaling-multimodal-llm-with-co-upcycled#ran","syntology_url":"https://syntology.ai/paper/2405.05949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.05949"}}}},{"paper":"/paper/self-supervised-visual-preference-alignment","slug":"self-supervised-visual-preference-alignment","title":"Self-Supervised Visual Preference Alignment","date":"2024-04-16","arxiv_id":"2404.10501","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":10,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":4,"samples_unverified":1,"pointer_only_for_licence":11,"official":{"repos":["Kevinz-code/SeVa"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/self-supervised-visual-preference-alignment#ran","syntology_url":"https://syntology.ai/paper/2404.10501","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10501"}}}},{"paper":"/paper/ferret-v2-an-improved-baseline-for-referring","slug":"ferret-v2-an-improved-baseline-for-referring","title":"Ferret-v2: An Improved Baseline for Referring and Grounding with Large Language Models","date":"2024-04-11","arxiv_id":"2404.07973","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":4,"samples_unverified":1,"pointer_only_for_licence":5,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/ferret-v2-an-improved-baseline-for-referring#ran","syntology_url":"https://syntology.ai/paper/2404.07973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07973"}}}},{"paper":"/paper/beyond-embeddings-the-promise-of-visual-table","slug":"beyond-embeddings-the-promise-of-visual-table","title":"Beyond Embeddings: The Promise of Visual Table in Visual Reasoning","date":"2024-03-27","arxiv_id":"2403.18252","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":6,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["lavi-lab/visual-table"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/beyond-embeddings-the-promise-of-visual-table#ran","syntology_url":"https://syntology.ai/paper/2403.18252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18252"}}}},{"paper":"/paper/mini-gemini-mining-the-potential-of-multi","slug":"mini-gemini-mining-the-potential-of-multi","title":"Mini-Gemini: Mining the Potential of Multi-modality Vision Language Models","date":"2024-03-27","arxiv_id":"2403.18814","rows_on_this_dataset":3,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":8,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["dvlab-research/minigemini"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mini-gemini-mining-the-potential-of-multi#ran","syntology_url":"https://syntology.ai/paper/2403.18814","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18814"}}}},{"paper":"/paper/chain-of-spot-interactive-reasoning-improves","slug":"chain-of-spot-interactive-reasoning-improves","title":"Chain-of-Spot: Interactive Reasoning Improves Large Vision-Language Models","date":"2024-03-19","arxiv_id":"2403.12966","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["dongyh20/chain-of-spot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/chain-of-spot-interactive-reasoning-improves#ran","syntology_url":"https://syntology.ai/paper/2403.12966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12966"}}}},{"paper":"/paper/sq-llava-self-questioning-for-large-vision","slug":"sq-llava-self-questioning-for-large-vision","title":"SQ-LLaVA: Self-Questioning for Large Vision-Language Assistant","date":"2024-03-17","arxiv_id":"2403.11299","rows_on_this_dataset":2,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":0,"official":{"repos":["heliossun/sq-llava"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/sq-llava-self-questioning-for-large-vision#ran","syntology_url":"https://syntology.ai/paper/2403.11299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11299"}}}},{"paper":"/paper/moai-mixture-of-all-intelligence-for-large","slug":"moai-mixture-of-all-intelligence-for-large","title":"MoAI: Mixture of All Intelligence for Large Language and Vision Models","date":"2024-03-12","arxiv_id":"2403.07508","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":1,"official":{"repos":["ByungKwanLee/MoAI"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/moai-mixture-of-all-intelligence-for-large#ran","syntology_url":"https://syntology.ai/paper/2403.07508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07508"}}}},{"paper":"/paper/deepseek-vl-towards-real-world-vision","slug":"deepseek-vl-towards-real-world-vision","title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","date":"2024-03-08","arxiv_id":"2403.05525","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":11,"samples_ran":11,"samples_constructed":0,"samples_ran_checked":7,"samples_ran_instrument_failed":4,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["deepseek-ai/deepseek-vl"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/deepseek-vl-towards-real-world-vision#ran","syntology_url":"https://syntology.ai/paper/2403.05525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05525"}}}},{"paper":"/paper/feast-your-eyes-mixture-of-resolution","slug":"feast-your-eyes-mixture-of-resolution","title":"Feast Your Eyes: Mixture-of-Resolution Adaptation for Multimodal Large Language Models","date":"2024-03-05","arxiv_id":"2403.03003","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["luogen1996/llava-hr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/feast-your-eyes-mixture-of-resolution#ran","syntology_url":"https://syntology.ai/paper/2403.03003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03003"}}}},{"paper":"/paper/the-all-seeing-project-v2-towards-general","slug":"the-all-seeing-project-v2-towards-general","title":"The All-Seeing Project V2: Towards General Relation Comprehension of the Open World","date":"2024-02-29","arxiv_id":"2402.19474","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":7,"samples_constructed":2,"samples_ran_checked":5,"samples_ran_instrument_failed":2,"samples_unverified":1,"pointer_only_for_licence":8,"official":{"repos":["opengvlab/all-seeing"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/the-all-seeing-project-v2-towards-general#ran","syntology_url":"https://syntology.ai/paper/2402.19474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.19474"}}}},{"paper":"/paper/tinyllava-a-framework-of-small-scale-large","slug":"tinyllava-a-framework-of-small-scale-large","title":"TinyLLaVA: A Framework of Small-scale Large Multimodal Models","date":"2024-02-22","arxiv_id":"2402.14289","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":4,"samples_unverified":4,"pointer_only_for_licence":1,"official":{"repos":["dlcv-buaa/tinyllavabench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/tinyllava-a-framework-of-small-scale-large#ran","syntology_url":"https://syntology.ai/paper/2402.14289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14289"}}}},{"paper":"/paper/collavo-crayon-large-language-and-vision","slug":"collavo-crayon-large-language-and-vision","title":"CoLLaVO: Crayon Large Language and Vision mOdel","date":"2024-02-17","arxiv_id":"2402.11248","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["ByungKwanLee/CoLLaVO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/collavo-crayon-large-language-and-vision#ran","syntology_url":"https://syntology.ai/paper/2402.11248","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11248"}}}},{"paper":"/paper/sphinx-x-scaling-data-and-parameters-for-a","slug":"sphinx-x-scaling-data-and-parameters-for-a","title":"SPHINX-X: Scaling Data and Parameters for a Family of Multi-modal Large Language Models","date":"2024-02-08","arxiv_id":"2402.05935","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":3,"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/sphinx-x-scaling-data-and-parameters-for-a#ran","syntology_url":"https://syntology.ai/paper/2402.05935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05935"}}}},{"paper":"/paper/video-lavit-unified-video-language-pre","slug":"video-lavit-unified-video-language-pre","title":"Video-LaVIT: Unified Video-Language Pre-training with Decoupled Visual-Motional Tokenization","date":"2024-02-05","arxiv_id":"2402.03161","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":2,"pointer_only_for_licence":5,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/video-lavit-unified-video-language-pre#ran","syntology_url":"https://syntology.ai/paper/2402.03161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03161"}}}},{"paper":"/paper/llava-ph-efficient-multi-modal-assistant-with","slug":"llava-ph-efficient-multi-modal-assistant-with","title":"LLaVA-Phi: Efficient Multi-Modal Assistant with Small Language Model","date":"2024-01-04","arxiv_id":"2401.02330","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":3,"samples_unverified":3,"pointer_only_for_licence":9,"official":{"repos":["zhuyiche/llava-phi"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/llava-ph-efficient-multi-modal-assistant-with#ran","syntology_url":"https://syntology.ai/paper/2401.02330","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02330"}}}},{"paper":"/paper/textit-v-guided-visual-search-as-a-core","slug":"textit-v-guided-visual-search-as-a-core","title":"V*: Guided Visual Search as a Core Mechanism in Multimodal LLMs","date":"2023-12-21","arxiv_id":"2312.14135","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["penghao-wu/vstar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/textit-v-guided-visual-search-as-a-core#ran","syntology_url":"https://syntology.ai/paper/2312.14135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14135"}}}},{"paper":"/paper/generative-multimodal-models-are-in-context","slug":"generative-multimodal-models-are-in-context","title":"Generative Multimodal Models are In-Context Learners","date":"2023-12-20","arxiv_id":"2312.13286","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":0,"samples_unverified":1,"pointer_only_for_licence":0,"official":{"repos":["baaivision/emu"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/generative-multimodal-models-are-in-context#ran","syntology_url":"https://syntology.ai/paper/2312.13286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13286"}}}},{"paper":"/paper/cogagent-a-visual-language-model-for-gui","slug":"cogagent-a-visual-language-model-for-gui","title":"CogAgent: A Visual Language Model for GUI Agents","date":"2023-12-14","arxiv_id":"2312.08914","rows_on_this_dataset":1,"code_links":3,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":18,"samples_ran":12,"samples_constructed":0,"samples_ran_checked":12,"samples_ran_instrument_failed":0,"samples_unverified":6,"pointer_only_for_licence":1,"official":{"repos":["THUDM/CogAgent","thudm/cogvlm"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":6,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/cogagent-a-visual-language-model-for-gui#ran","syntology_url":"https://syntology.ai/paper/2312.08914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.08914"}}}},{"paper":"/paper/hallucination-augmented-contrastive-learning","slug":"hallucination-augmented-contrastive-learning","title":"Hallucination Augmented Contrastive Learning for Multimodal Large Language Model","date":"2023-12-12","arxiv_id":"2312.06968","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":6,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":3,"pointer_only_for_licence":5,"official":{"repos":["x-plug/mplug-halowl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/hallucination-augmented-contrastive-learning#ran","syntology_url":"https://syntology.ai/paper/2312.06968","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06968"}}}},{"paper":"/paper/onellm-one-framework-to-align-all-modalities","slug":"onellm-one-framework-to-align-all-modalities","title":"OneLLM: One Framework to Align All Modalities with Language","date":"2023-12-06","arxiv_id":"2312.03700","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":2,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":4,"official":{"repos":["csuhan/onellm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/onellm-one-framework-to-align-all-modalities#ran","syntology_url":"https://syntology.ai/paper/2312.03700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03700"}}}},{"paper":"/paper/video-llava-learning-united-visual-1","slug":"video-llava-learning-united-visual-1","title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","date":"2023-11-16","arxiv_id":"2311.10122","rows_on_this_dataset":1,"code_links":6,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":2,"pointer_only_for_licence":2,"official":{"repos":["PKU-YuanGroup/Video-LLaVA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/video-llava-learning-united-visual-1#ran","syntology_url":"https://syntology.ai/paper/2311.10122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10122"}}}},{"paper":"/paper/volcano-mitigating-multimodal-hallucination","slug":"volcano-mitigating-multimodal-hallucination","title":"Volcano: Mitigating Multimodal Hallucination through Self-Feedback Guided Revision","date":"2023-11-13","arxiv_id":"2311.07362","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":10,"samples_ran":9,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":4,"samples_unverified":1,"pointer_only_for_licence":10,"official":{"repos":["kaistai/volcano"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/volcano-mitigating-multimodal-hallucination#ran","syntology_url":"https://syntology.ai/paper/2311.07362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07362"}}}},{"paper":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and","slug":"sphinx-the-joint-mixing-of-weights-tasks-and","title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","date":"2023-11-13","arxiv_id":"2311.07575","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":3,"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and#ran","syntology_url":"https://syntology.ai/paper/2311.07575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07575"}}}},{"paper":"/paper/llava-plus-learning-to-use-tools-for-creating","slug":"llava-plus-learning-to-use-tools-for-creating","title":"LLaVA-Plus: Learning to Use Tools for Creating Multimodal Agents","date":"2023-11-09","arxiv_id":"2311.05437","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":2,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["LLaVA-VL/LLaVA-Plus-Codebase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/llava-plus-learning-to-use-tools-for-creating#ran","syntology_url":"https://syntology.ai/paper/2311.05437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05437"}}}},{"paper":"/paper/otterhd-a-high-resolution-multi-modality","slug":"otterhd-a-high-resolution-multi-modality","title":"OtterHD: A High-Resolution Multi-modality Model","date":"2023-11-07","arxiv_id":"2311.04219","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":14,"samples_ran":13,"samples_constructed":0,"samples_ran_checked":12,"samples_ran_instrument_failed":1,"samples_unverified":1,"pointer_only_for_licence":3,"official":{"repos":["luodian/otter"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/otterhd-a-high-resolution-multi-modality#ran","syntology_url":"https://syntology.ai/paper/2311.04219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04219"}}}},{"paper":"/paper/mplug-owl2-revolutionizing-multi-modal-large","slug":"mplug-owl2-revolutionizing-multi-modal-large","title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","date":"2023-11-07","arxiv_id":"2311.04257","rows_on_this_dataset":1,"code_links":2,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":1,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":1,"samples_unverified":2,"pointer_only_for_licence":3,"official":{"repos":["x-plug/mplug-owl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mplug-owl2-revolutionizing-multi-modal-large#ran","syntology_url":"https://syntology.ai/paper/2311.04257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04257"}}}},{"paper":"/paper/improved-baselines-with-visual-instruction","slug":"improved-baselines-with-visual-instruction","title":"Improved Baselines with Visual Instruction Tuning","date":"2023-10-05","arxiv_id":"2310.03744","rows_on_this_dataset":2,"code_links":9,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":9,"samples_ran":6,"samples_constructed":0,"samples_ran_checked":3,"samples_ran_instrument_failed":3,"samples_unverified":3,"pointer_only_for_licence":8,"official":null,"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/improved-baselines-with-visual-instruction#ran","syntology_url":"https://syntology.ai/paper/2310.03744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03744"}}}},{"paper":"/paper/dreamllm-synergistic-multimodal-comprehension","slug":"dreamllm-synergistic-multimodal-comprehension","title":"DreamLLM: Synergistic Multimodal Comprehension and Creation","date":"2023-09-20","arxiv_id":"2309.11499","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":3,"samples_ran":3,"samples_constructed":0,"samples_ran_checked":0,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["RunpeiDong/DreamLLM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/dreamllm-synergistic-multimodal-comprehension#ran","syntology_url":"https://syntology.ai/paper/2309.11499","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11499"}}}},{"paper":"/paper/an-empirical-study-of-scaling-instruct-tuned","slug":"an-empirical-study-of-scaling-instruct-tuned","title":"An Empirical Study of Scaling Instruct-Tuned Large Multimodal Models","date":"2023-09-18","arxiv_id":"2309.09958","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":2,"samples_ran":2,"samples_constructed":0,"samples_ran_checked":1,"samples_ran_instrument_failed":1,"samples_unverified":0,"pointer_only_for_licence":2,"official":{"repos":["haotian-liu/LLaVA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/an-empirical-study-of-scaling-instruct-tuned#ran","syntology_url":"https://syntology.ai/paper/2309.09958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.09958"}}}},{"paper":"/paper/stablellava-enhanced-visual-instruction","slug":"stablellava-enhanced-visual-instruction","title":"StableLLaVA: Enhanced Visual Instruction Tuning with Synthesized Image-Dialogue Data","date":"2023-08-20","arxiv_id":"2308.10253","rows_on_this_dataset":1,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":7,"samples_ran":7,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":3,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["icoz69/stablellava"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/stablellava-enhanced-visual-instruction#ran","syntology_url":"https://syntology.ai/paper/2308.10253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10253"}}}},{"paper":"/paper/mm-react-prompting-chatgpt-for-multimodal","slug":"mm-react-prompting-chatgpt-for-multimodal","title":"MM-REACT: Prompting ChatGPT for Multimodal Reasoning and Action","date":"2023-03-20","arxiv_id":"2303.11381","rows_on_this_dataset":2,"code_links":1,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":4,"samples_ran":4,"samples_constructed":0,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":0,"official":{"repos":["microsoft/MM-REACT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/mm-react-prompting-chatgpt-for-multimodal#ran","syntology_url":"https://syntology.ai/paper/2303.11381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11381"}}}},{"paper":"/paper/gpt-4-technical-report-1","slug":"gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","rows_on_this_dataset":5,"code_links":11,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":5,"samples_ran":5,"samples_constructed":0,"samples_ran_checked":5,"samples_ran_instrument_failed":0,"samples_unverified":0,"pointer_only_for_licence":1,"official":{"repos":["openai/evals"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/gpt-4-technical-report-1#ran","syntology_url":"https://syntology.ai/paper/2303.08774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08774"}}}},{"paper":"/paper/blip-2-bootstrapping-language-image-pre","slug":"blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","rows_on_this_dataset":1,"code_links":17,"syntology":{"read_at":"2026-09-28T10:30:06+00:00","samples_harvested":8,"samples_ran":4,"samples_constructed":4,"samples_ran_checked":4,"samples_ran_instrument_failed":0,"samples_unverified":4,"pointer_only_for_licence":1,"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]},"claim":"Per-sample execution on synthesized fixtures; not a correctness claim.","sample_list":"/paper/blip-2-bootstrapping-language-image-pre#ran","syntology_url":"https://syntology.ai/paper/2301.12597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12597"}}}}],"record_sha256":"b32333e3fe2208094d4f5abc1c0fab3d8646d9925171a0029baa8dd3b050700e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}