{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/multimodal-reasoning/papers/ran/1","list_of":"/task/multimodal-reasoning","task":"Multimodal Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,52],"of":52,"counts":{"archive_papers_tagged":302,"with_a_code_link":138,"where_syntology_ran_a_sample":52,"not_listed_spam_title":0,"listed":302,"listed_where_code_ran":52,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":43,"every_run_a_failure_of_syntologys_instrument":9,"listed_with_a_run_with_no_instrument_failure":43,"listed_every_run_a_failure_of_syntologys_instrument":9,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/multimodal-reasoning/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/skywork-r1v3-technical-report","slug":"skywork-r1v3-technical-report","title":"Skywork-R1V3 Technical Report","date":"2025-07-08","arxiv_id":"2507.06167","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/skywork-r1v3-technical-report#ran","syntology_url":"https://syntology.ai/paper/2507.06167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.06167"}},"official":{"repos":["SkyworkAI/Skywork-R1V"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/dreamvla-a-vision-language-action-model-1","slug":"dreamvla-a-vision-language-action-model-1","title":"DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge","date":"2025-07-06","arxiv_id":"2507.04447","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/dreamvla-a-vision-language-action-model-1#ran","syntology_url":"https://syntology.ai/paper/2507.04447","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.04447"}},"official":{"repos":["Zhangwenyao1/DreamVLA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/glm-4-1v-thinking-towards-versatile","slug":"glm-4-1v-thinking-towards-versatile","title":"GLM-4.1V-Thinking: Towards Versatile Multimodal Reasoning with Scalable Reinforcement Learning","date":"2025-07-01","arxiv_id":"2507.01006","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glm-4-1v-thinking-towards-versatile#ran","syntology_url":"https://syntology.ai/paper/2507.01006","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.01006"}},"official":{"repos":["thudm/glm-4.1v-thinking"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/machine-mental-imagery-empower-multimodal","slug":"machine-mental-imagery-empower-multimodal","title":"Machine Mental Imagery: Empower Multimodal Reasoning with Latent Visual Tokens","date":"2025-06-20","arxiv_id":"2506.17218","repositories_listed":1,"syntology":{"n":17,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":17,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/machine-mental-imagery-empower-multimodal#ran","syntology_url":"https://syntology.ai/paper/2506.17218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.17218"}},"official":{"repos":["umass-embodied-agi/mirage"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/play-to-generalize-learning-to-reason-through","slug":"play-to-generalize-learning-to-reason-through","title":"Play to Generalize: Learning to Reason Through Game Play","date":"2025-06-09","arxiv_id":"2506.08011","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/play-to-generalize-learning-to-reason-through#ran","syntology_url":"https://syntology.ai/paper/2506.08011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.08011"}},"official":{"repos":["yunfeixie233/vigal"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/mimo-vl-technical-report","slug":"mimo-vl-technical-report","title":"MiMo-VL Technical Report","date":"2025-06-04","arxiv_id":"2506.03569","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mimo-vl-technical-report#ran","syntology_url":"https://syntology.ai/paper/2506.03569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.03569"}},"official":{"repos":["xiaomimimo/mimo-vl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vidtext-towards-comprehensive-evaluation-for","slug":"vidtext-towards-comprehensive-evaluation-for","title":"VidText: Towards Comprehensive Evaluation for Video Text Understanding","date":"2025-05-28","arxiv_id":"2505.22810","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vidtext-towards-comprehensive-evaluation-for#ran","syntology_url":"https://syntology.ai/paper/2505.22810","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.22810"}},"official":{"repos":["shuyansy/vidtext"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vtool-r1-vlms-learn-to-think-with-images-via","slug":"vtool-r1-vlms-learn-to-think-with-images-via","title":"VTool-R1: VLMs Learn to Think with Images via Reinforcement Learning on Multimodal Tool Use","date":"2025-05-25","arxiv_id":"2505.19255","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vtool-r1-vlms-learn-to-think-with-images-via#ran","syntology_url":"https://syntology.ai/paper/2505.19255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19255"}},"official":null}},{"url":"/paper/chartgalaxy-a-dataset-for-infographic-chart","slug":"chartgalaxy-a-dataset-for-infographic-chart","title":"ChartGalaxy: A Dataset for Infographic Chart Understanding and Generation","date":"2025-05-24","arxiv_id":"2505.18668","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chartgalaxy-a-dataset-for-infographic-chart#ran","syntology_url":"https://syntology.ai/paper/2505.18668","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.18668"}},"official":{"repos":["chartgalaxy/chartgalaxy"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/code2logic-game-code-driven-data-synthesis","slug":"code2logic-game-code-driven-data-synthesis","title":"Code2Logic: Game-Code-Driven Data Synthesis for Enhancing VLMs General Reasoning","date":"2025-05-20","arxiv_id":"2505.13886","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":15,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/code2logic-game-code-driven-data-synthesis#ran","syntology_url":"https://syntology.ai/paper/2505.13886","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13886"}},"official":{"repos":["tongjingqi/code2logic"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepeyes-incentivizing-thinking-with-images","slug":"deepeyes-incentivizing-thinking-with-images","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","date":"2025-05-20","arxiv_id":"2505.14362","repositories_listed":1,"syntology":{"n":17,"n_ran":12,"n_constructed":0,"n_ran_checked":9,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":4,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/deepeyes-incentivizing-thinking-with-images#ran","syntology_url":"https://syntology.ai/paper/2505.14362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14362"}},"official":{"repos":["visual-agent/deepeyes"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/emerging-properties-in-unified-multimodal","slug":"emerging-properties-in-unified-multimodal","title":"Emerging Properties in Unified Multimodal Pretraining","date":"2025-05-20","arxiv_id":"2505.14683","repositories_listed":2,"syntology":{"n":21,"n_ran":19,"n_constructed":0,"n_ran_checked":16,"n_instrument":3,"n_unverified":2,"n_honours":3,"n_violates":0,"n_no_contract":13,"n_pointer_only":3,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 3 honoured, 0 violated, 13 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/emerging-properties-in-unified-multimodal#ran","syntology_url":"https://syntology.ai/paper/2505.14683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.14683"}},"official":null}},{"url":"/paper/sakura-on-the-multi-hop-reasoning-of-large","slug":"sakura-on-the-multi-hop-reasoning-of-large","title":"SAKURA: On the Multi-hop Reasoning of Large Audio-Language Models Based on Speech and Audio Information","date":"2025-05-19","arxiv_id":"2505.13237","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sakura-on-the-multi-hop-reasoning-of-large#ran","syntology_url":"https://syntology.ai/paper/2505.13237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13237"}},"official":{"repos":["b08202033/SAKURA","ckyang1124/sakura"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-prm-enhancing-multimodal-mathematical","slug":"mm-prm-enhancing-multimodal-mathematical","title":"MM-PRM: Enhancing Multimodal Mathematical Reasoning with Scalable Step-Level Supervision","date":"2025-05-19","arxiv_id":"2505.13427","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mm-prm-enhancing-multimodal-mathematical#ran","syntology_url":"https://syntology.ai/paper/2505.13427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13427"}},"official":{"repos":["modalminds/mm-prm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/skywork-r1v2-multimodal-hybrid-reinforcement","slug":"skywork-r1v2-multimodal-hybrid-reinforcement","title":"Skywork R1V2: Multimodal Hybrid Reinforcement Learning for Reasoning","date":"2025-04-23","arxiv_id":"2504.16656","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/skywork-r1v2-multimodal-hybrid-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2504.16656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16656"}},"official":{"repos":["SkyworkAI/Skywork-R1V"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-rethinker-incentivizing-self-reflection-of","slug":"vl-rethinker-incentivizing-self-reflection-of","title":"VL-Rethinker: Incentivizing Self-Reflection of Vision-Language Models with Reinforcement Learning","date":"2025-04-10","arxiv_id":"2504.08837","repositories_listed":2,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vl-rethinker-incentivizing-self-reflection-of#ran","syntology_url":"https://syntology.ai/paper/2504.08837","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.08837"}},"official":null}},{"url":"/paper/skywork-r1v-pioneering-multimodal-reasoning","slug":"skywork-r1v-pioneering-multimodal-reasoning","title":"Skywork R1V: Pioneering Multimodal Reasoning with Chain-of-Thought","date":"2025-04-08","arxiv_id":"2504.05599","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/skywork-r1v-pioneering-multimodal-reasoning#ran","syntology_url":"https://syntology.ai/paper/2504.05599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05599"}},"official":null}},{"url":"/paper/3mdbench-medical-multimodal-multi-agent","slug":"3mdbench-medical-multimodal-multi-agent","title":"3MDBench: Medical Multimodal Multi-agent Dialogue Benchmark","date":"2025-03-26","arxiv_id":"2504.13861","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/3mdbench-medical-multimodal-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2504.13861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13861"}},"official":{"repos":["univanxx/3mdbench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/openvlthinker-an-early-exploration-to-complex","slug":"openvlthinker-an-early-exploration-to-complex","title":"OpenVLThinker: An Early Exploration to Complex Vision-Language Reasoning via Iterative Self-Improvement","date":"2025-03-21","arxiv_id":"2503.17352","repositories_listed":1,"syntology":{"n":21,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":12,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/openvlthinker-an-early-exploration-to-complex#ran","syntology_url":"https://syntology.ai/paper/2503.17352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.17352"}},"official":{"repos":["yihedeng9/openvlthinker"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":12,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-more-a-comparative-study-of-llms-and","slug":"llava-more-a-comparative-study-of-llms-and","title":"LLaVA-MORE: A Comparative Study of LLMs and Visual Backbones for Enhanced Visual Instruction Tuning","date":"2025-03-19","arxiv_id":"2503.15621","repositories_listed":1,"syntology":{"n":14,"n_ran":10,"n_constructed":0,"n_ran_checked":7,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":2,"n_no_contract":5,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 2 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/llava-more-a-comparative-study-of-llms-and#ran","syntology_url":"https://syntology.ai/paper/2503.15621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.15621"}},"official":{"repos":["aimagelab/LLaVA-MORE"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/deepperception-advancing-r1-like-cognitive","slug":"deepperception-advancing-r1-like-cognitive","title":"DeepPerception: Advancing R1-like Cognitive Visual Perception in MLLMs for Knowledge-Intensive Visual Grounding","date":"2025-03-17","arxiv_id":"2503.12797","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deepperception-advancing-r1-like-cognitive#ran","syntology_url":"https://syntology.ai/paper/2503.12797","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.12797"}},"official":{"repos":["thunlp/deepperception"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/r1-onevision-advancing-generalized-multimodal","slug":"r1-onevision-advancing-generalized-multimodal","title":"R1-Onevision: Advancing Generalized Multimodal Reasoning through Cross-Modal Formalization","date":"2025-03-13","arxiv_id":"2503.10615","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r1-onevision-advancing-generalized-multimodal#ran","syntology_url":"https://syntology.ai/paper/2503.10615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10615"}},"official":{"repos":["Fancy-MLLM/R1-onevision"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mindgym-enhancing-vision-language-models-via","slug":"mindgym-enhancing-vision-language-models-via","title":"MindGYM: Enhancing Vision-Language Models via Synthetic Self-Challenging Questions","date":"2025-03-12","arxiv_id":"2503.09499","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mindgym-enhancing-vision-language-models-via#ran","syntology_url":"https://syntology.ai/paper/2503.09499","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.09499"}},"official":{"repos":["modelscope/data-juicer"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/oasis-one-image-is-all-you-need-for","slug":"oasis-one-image-is-all-you-need-for","title":"Oasis: One Image is All You Need for Multimodal Instruction Data Synthesis","date":"2025-03-11","arxiv_id":"2503.08741","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/oasis-one-image-is-all-you-need-for#ran","syntology_url":"https://syntology.ai/paper/2503.08741","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08741"}},"official":{"repos":["Letian2003/MM_INF"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lmm-r1-empowering-3b-lmms-with-strong","slug":"lmm-r1-empowering-3b-lmms-with-strong","title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","date":"2025-03-10","arxiv_id":"2503.07536","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lmm-r1-empowering-3b-lmms-with-strong#ran","syntology_url":"https://syntology.ai/paper/2503.07536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07536"}},"official":null}},{"url":"/paper/can-atomic-step-decomposition-enhance-the","slug":"can-atomic-step-decomposition-enhance-the","title":"Can Atomic Step Decomposition Enhance the Self-structured Reasoning of Multimodal Large Models?","date":"2025-03-08","arxiv_id":"2503.06252","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/can-atomic-step-decomposition-enhance-the#ran","syntology_url":"https://syntology.ai/paper/2503.06252","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06252"}},"official":{"repos":["quinn777/atomthink"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a","slug":"r1-zero-s-aha-moment-in-visual-reasoning-on-a","title":"R1-Zero's \"Aha Moment\" in Visual Reasoning on a 2B Non-SFT Model","date":"2025-03-07","arxiv_id":"2503.05132","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a#ran","syntology_url":"https://syntology.ai/paper/2503.05132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05132"}},"official":{"repos":["turningpoint-ai/visualthinker-r1-zero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/audio-reasoner-improving-reasoning-capability","slug":"audio-reasoner-improving-reasoning-capability","title":"Audio-Reasoner: Improving Reasoning Capability in Large Audio Language Models","date":"2025-03-04","arxiv_id":"2503.02318","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/audio-reasoner-improving-reasoning-capability#ran","syntology_url":"https://syntology.ai/paper/2503.02318","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.02318"}},"official":null}},{"url":"/paper/efficient-reasoning-with-hidden-thinking","slug":"efficient-reasoning-with-hidden-thinking","title":"Efficient Reasoning with Hidden Thinking","date":"2025-01-31","arxiv_id":"2501.19201","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-reasoning-with-hidden-thinking#ran","syntology_url":"https://syntology.ai/paper/2501.19201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19201"}},"official":{"repos":["shawnricecake/heima"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-o1-let-vision-language-models-reason","slug":"llava-o1-let-vision-language-models-reason","title":"LLaVA-CoT: Let Vision Language Models Reason Step-by-Step","date":"2024-11-15","arxiv_id":"2411.10440","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-o1-let-vision-language-models-reason#ran","syntology_url":"https://syntology.ai/paper/2411.10440","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.10440"}},"official":{"repos":["PKU-YuanGroup/LLaVA-CoT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-low-resource-harmful-meme-detection","slug":"towards-low-resource-harmful-meme-detection","title":"Towards Low-Resource Harmful Meme Detection with LMM Agents","date":"2024-11-08","arxiv_id":"2411.05383","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-low-resource-harmful-meme-detection#ran","syntology_url":"https://syntology.ai/paper/2411.05383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05383"}},"official":{"repos":["jianzhao-huang/lorehm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/distill-visual-chart-reasoning-ability-from","slug":"distill-visual-chart-reasoning-ability-from","title":"Distill Visual Chart Reasoning Ability from LLMs to MLLMs","date":"2024-10-24","arxiv_id":"2410.18798","repositories_listed":2,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/distill-visual-chart-reasoning-ability-from#ran","syntology_url":"https://syntology.ai/paper/2410.18798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18798"}},"official":{"repos":["hewei2001/reachqa"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/journeybench-a-challenging-one-stop-vision","slug":"journeybench-a-challenging-one-stop-vision","title":"JourneyBench: A Challenging One-Stop Vision-Language Understanding Benchmark of Generated Images","date":"2024-09-19","arxiv_id":"2409.12953","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/journeybench-a-challenging-one-stop-vision#ran","syntology_url":"https://syntology.ai/paper/2409.12953","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.12953"}},"official":{"repos":["journeybench/journeybench"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/math-puma-progressive-upward-multimodal","slug":"math-puma-progressive-upward-multimodal","title":"Math-PUMA: Progressive Upward Multimodal Alignment to Enhance Mathematical Reasoning","date":"2024-08-16","arxiv_id":"2408.08640","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/math-puma-progressive-upward-multimodal#ran","syntology_url":"https://syntology.ai/paper/2408.08640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08640"}},"official":{"repos":["wwzhuang01/math-puma"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2408-01337","slug":"2408-01337","title":"MuChoMusic: Evaluating Music Understanding in Multimodal Audio-Language Models","date":"2024-08-02","arxiv_id":"2408.01337","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2408-01337#ran","syntology_url":"https://syntology.ai/paper/2408.01337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.01337"}},"official":{"repos":["mulab-mir/muchomusic"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cofipara-a-coarse-to-fine-paradigm-for","slug":"cofipara-a-coarse-to-fine-paradigm-for","title":"CofiPara: A Coarse-to-fine Paradigm for Multimodal Sarcasm Target Identification with Large Multimodal Models","date":"2024-05-01","arxiv_id":"2405.00390","repositories_listed":1,"syntology":{"n":21,"n_ran":18,"n_constructed":0,"n_ran_checked":18,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":18,"n_pointer_only":21,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cofipara-a-coarse-to-fine-paradigm-for#ran","syntology_url":"https://syntology.ai/paper/2405.00390","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.00390"}},"official":{"repos":["lbotirx/cofipara"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/puzzlevqa-diagnosing-multimodal-reasoning","slug":"puzzlevqa-diagnosing-multimodal-reasoning","title":"PuzzleVQA: Diagnosing Multimodal Reasoning Challenges of Language Models with Abstract Visual Patterns","date":"2024-03-20","arxiv_id":"2403.13315","repositories_listed":2,"syntology":{"n":14,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":1,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/puzzlevqa-diagnosing-multimodal-reasoning#ran","syntology_url":"https://syntology.ai/paper/2403.13315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13315"}},"official":{"repos":["declare-lab/llm-puzzletest"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/measuring-vision-language-stem-skills-of","slug":"measuring-vision-language-stem-skills-of","title":"Measuring Vision-Language STEM Skills of Neural Models","date":"2024-02-27","arxiv_id":"2402.17205","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-vision-language-stem-skills-of#ran","syntology_url":"https://syntology.ai/paper/2402.17205","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17205"}},"official":{"repos":["stemdataset/STEM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stop-reasoning-when-multimodal-llms-with","slug":"stop-reasoning-when-multimodal-llms-with","title":"Stop Reasoning! When Multimodal LLM with Chain-of-Thought Reasoning Meets Adversarial Image","date":"2024-02-22","arxiv_id":"2402.14899","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stop-reasoning-when-multimodal-llms-with#ran","syntology_url":"https://syntology.ai/paper/2402.14899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14899"}},"official":{"repos":["aipenguin/stopreasoning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/crema-multimodal-compositional-video","slug":"crema-multimodal-compositional-video","title":"CREMA: Generalizable and Efficient Video-Language Reasoning via Multimodal Modular Fusion","date":"2024-02-08","arxiv_id":"2402.05889","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crema-multimodal-compositional-video#ran","syntology_url":"https://syntology.ai/paper/2402.05889","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.05889"}},"official":{"repos":["Yui010206/CREMA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-generalization-capacity-of-neural","slug":"on-the-generalization-capacity-of-neural","title":"On the generalization capacity of neural networks during generic multimodal reasoning","date":"2024-01-26","arxiv_id":"2401.15030","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-the-generalization-capacity-of-neural#ran","syntology_url":"https://syntology.ai/paper/2401.15030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.15030"}},"official":{"repos":["ibm/gcog"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/boosting-the-power-of-small-multimodal","slug":"boosting-the-power-of-small-multimodal","title":"Boosting the Power of Small Multimodal Reasoning Models to Match Larger Models with Self-Consistency Training","date":"2023-11-23","arxiv_id":"2311.14109","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-the-power-of-small-multimodal#ran","syntology_url":"https://syntology.ai/paper/2311.14109","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.14109"}},"official":{"repos":["chengtan9907/mc-cot"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-human-like-multi-modal-reasoning-a","slug":"enhancing-human-like-multi-modal-reasoning-a","title":"Enhancing Human-like Multi-Modal Reasoning: A New Challenging Dataset and Comprehensive Framework","date":"2023-07-24","arxiv_id":"2307.12626","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":3,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/enhancing-human-like-multi-modal-reasoning-a#ran","syntology_url":"https://syntology.ai/paper/2307.12626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12626"}},"official":{"repos":["weijingxuan/COCO-MMR"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-chain-of-thought-effective-graph-of","slug":"beyond-chain-of-thought-effective-graph-of","title":"Beyond Chain-of-Thought, Effective Graph-of-Thought Reasoning in Language Models","date":"2023-05-26","arxiv_id":"2305.16582","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/beyond-chain-of-thought-effective-graph-of#ran","syntology_url":"https://syntology.ai/paper/2305.16582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16582"}},"official":{"repos":["zoeyyao27/graph-of-thought"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-itself-can-read-and-generate-cxr-images","slug":"llm-itself-can-read-and-generate-cxr-images","title":"LLM-CXR: Instruction-Finetuned LLM for CXR Image Understanding and Generation","date":"2023-05-19","arxiv_id":"2305.11490","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/llm-itself-can-read-and-generate-cxr-images#ran","syntology_url":"https://syntology.ai/paper/2305.11490","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11490"}},"official":{"repos":["hyn2028/llm-cxr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-react-prompting-chatgpt-for-multimodal","slug":"mm-react-prompting-chatgpt-for-multimodal","title":"MM-REACT: Prompting ChatGPT for Multimodal Reasoning and Action","date":"2023-03-20","arxiv_id":"2303.11381","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mm-react-prompting-chatgpt-for-multimodal#ran","syntology_url":"https://syntology.ai/paper/2303.11381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11381"}},"official":{"repos":["microsoft/MM-REACT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multimodal-analogical-reasoning-over","slug":"multimodal-analogical-reasoning-over","title":"Multimodal Analogical Reasoning over Knowledge Graphs","date":"2022-10-01","arxiv_id":"2210.00312","repositories_listed":2,"syntology":{"n":27,"n_ran":17,"n_constructed":2,"n_ran_checked":14,"n_instrument":3,"n_unverified":10,"n_honours":1,"n_violates":1,"n_no_contract":12,"n_pointer_only":2,"phrase":"17 ran (of which 2 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 1 violated, 12 with no contract checked; 3 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/multimodal-analogical-reasoning-over#ran","syntology_url":"https://syntology.ai/paper/2210.00312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.00312"}},"official":{"repos":["zjunlp/MKG_Analogy"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":4,"n_unverified":10,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/learn-to-explain-multimodal-reasoning-via","slug":"learn-to-explain-multimodal-reasoning-via","title":"Learn to Explain: Multimodal Reasoning via Thought Chains for Science Question Answering","date":"2022-09-20","arxiv_id":"2209.09513","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learn-to-explain-multimodal-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2209.09513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.09513"}},"official":{"repos":["lupantech/ScienceQA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/do-vision-language-pretrained-models-learn","slug":"do-vision-language-pretrained-models-learn","title":"Do Vision-Language Pretrained Models Learn Composable Primitive Concepts?","date":"2022-03-31","arxiv_id":"2203.17271","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/do-vision-language-pretrained-models-learn#ran","syntology_url":"https://syntology.ai/paper/2203.17271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.17271"}},"official":{"repos":["tttyuntian/vlm_primitive_concepts"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/webqa-multihop-and-multimodal-qa","slug":"webqa-multihop-and-multimodal-qa","title":"WebQA: Multihop and Multimodal QA","date":"2021-09-01","arxiv_id":"2109.00590","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/webqa-multihop-and-multimodal-qa#ran","syntology_url":"https://syntology.ai/paper/2109.00590","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00590"}},"official":null}},{"url":"/paper/merlot-multimodal-neural-script-knowledge","slug":"merlot-multimodal-neural-script-knowledge","title":"MERLOT: Multimodal Neural Script Knowledge Models","date":"2021-06-04","arxiv_id":"2106.02636","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/merlot-multimodal-neural-script-knowledge#ran","syntology_url":"https://syntology.ai/paper/2106.02636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.02636"}},"official":null}},{"url":"/paper/e-snli-ve-2-0-corrected-visual-textual","slug":"e-snli-ve-2-0-corrected-visual-textual","title":"e-SNLI-VE: Corrected Visual-Textual Entailment with Natural Language Explanations","date":"2020-04-07","arxiv_id":"2004.03744","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/e-snli-ve-2-0-corrected-visual-textual#ran","syntology_url":"https://syntology.ai/paper/2004.03744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.03744"}},"official":{"repos":["virginie-do/e-SNLI-VE"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}}],"record_sha256":"584637fba0830f78e082253866a8aa199772ef837d197b63ec70612e223da989","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}