{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-reasoning/papers/3","list_of":"/task/visual-reasoning","task":"Visual Reasoning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":7,"rows_per_page":100,"rows":[201,300],"of":698,"counts":{"archive_papers_tagged":698,"with_a_code_link":356,"where_syntology_ran_a_sample":165,"not_listed_spam_title":0,"listed":698,"listed_where_code_ran":165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":130,"every_run_a_failure_of_syntologys_instrument":35,"listed_with_a_run_with_no_instrument_failure":130,"listed_every_run_a_failure_of_syntologys_instrument":35,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-reasoning","prev":"/task/visual-reasoning/papers/2","next":"/task/visual-reasoning/papers/4","papers":[{"url":"/paper/stop-reasoning-when-multimodal-llms-with","slug":"stop-reasoning-when-multimodal-llms-with","title":"Stop Reasoning! When Multimodal LLM with Chain-of-Thought Reasoning Meets Adversarial Image","date":"2024-02-22","arxiv_id":"2402.14899","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/stop-reasoning-when-multimodal-llms-with#ran","syntology_url":"https://syntology.ai/paper/2402.14899","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14899"}},"official":{"repos":["aipenguin/stopreasoning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-reasoning-in-object-centric-deep","slug":"visual-reasoning-in-object-centric-deep","title":"Visual Reasoning in Object-Centric Deep Neural Networks: A Comparative Cognition Approach","date":"2024-02-20","arxiv_id":"2402.12675","repositories_listed":1,"syntology":null},{"url":"/paper/vigor-improving-visual-grounding-of-large","slug":"vigor-improving-visual-grounding-of-large","title":"ViGoR: Improving Visual Grounding of Large Vision Language Models with Fine-Grained Reward Modeling","date":"2024-02-09","arxiv_id":"2402.06118","repositories_listed":1,"syntology":null},{"url":"/paper/cogcom-train-large-vision-language-models","slug":"cogcom-train-large-vision-language-models","title":"CogCoM: Train Large Vision-Language Models Diving into Details through Chain of Manipulations","date":"2024-02-06","arxiv_id":"2402.04236","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cogcom-train-large-vision-language-models#ran","syntology_url":"https://syntology.ai/paper/2402.04236","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04236"}},"official":{"repos":["thudm/cogcom"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contextual-evaluating-context-sensitive-text","slug":"contextual-evaluating-context-sensitive-text","title":"ConTextual: Evaluating Context-Sensitive Text-Rich Visual Reasoning in Large Multimodal Models","date":"2024-01-24","arxiv_id":"2401.13311","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contextual-evaluating-context-sensitive-text#ran","syntology_url":"https://syntology.ai/paper/2401.13311","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13311"}},"official":{"repos":["rohan598/contextual"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompting-large-vision-language-models-for","slug":"prompting-large-vision-language-models-for","title":"Prompting Large Vision-Language Models for Compositional Reasoning","date":"2024-01-20","arxiv_id":"2401.11337","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/prompting-large-vision-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2401.11337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11337"}},"official":{"repos":["tossowski/keycomp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/image-safeguarding-reasoning-with-conditional","slug":"image-safeguarding-reasoning-with-conditional","title":"Image Safeguarding: Reasoning with Conditional Vision Language Model and Obfuscating Unsafe Content Counterfactually","date":"2024-01-19","arxiv_id":"2401.11035","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/image-safeguarding-reasoning-with-conditional#ran","syntology_url":"https://syntology.ai/paper/2401.11035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.11035"}},"official":{"repos":["secureaiautonomylab/conditionalvlm"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-generative-abstract-reasoning","slug":"towards-generative-abstract-reasoning","title":"Towards Generative Abstract Reasoning: Completing Raven's Progressive Matrix via Rule Abstraction and Selection","date":"2024-01-18","arxiv_id":"2401.09966","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-generative-abstract-reasoning#ran","syntology_url":"https://syntology.ai/paper/2401.09966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09966"}},"official":{"repos":["fudanvi/generative-abstract-reasoning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cocot-contrastive-chain-of-thought-prompting","slug":"cocot-contrastive-chain-of-thought-prompting","title":"CoCoT: Contrastive Chain-of-Thought Prompting for Large Multimodal Models with Multiple Image Inputs","date":"2024-01-05","arxiv_id":"2401.02582","repositories_listed":1,"syntology":null},{"url":"/paper/vcoder-versatile-vision-encoders-for","slug":"vcoder-versatile-vision-encoders-for","title":"VCoder: Versatile Vision Encoders for Multimodal Large Language Models","date":"2023-12-21","arxiv_id":"2312.14233","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vcoder-versatile-vision-encoders-for#ran","syntology_url":"https://syntology.ai/paper/2312.14233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14233"}},"official":{"repos":["shi-labs/vcoder"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/one-self-configurable-model-to-solve-many","slug":"one-self-configurable-model-to-solve-many","title":"One Self-Configurable Model to Solve Many Abstract Visual Reasoning Problems","date":"2023-12-15","arxiv_id":"2312.09997","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/one-self-configurable-model-to-solve-many#ran","syntology_url":"https://syntology.ai/paper/2312.09997","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.09997"}},"official":{"repos":["mikomel/sal"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt4sgg-synthesizing-scene-graphs-from","slug":"gpt4sgg-synthesizing-scene-graphs-from","title":"GPT4SGG: Synthesizing Scene Graphs from Holistic and Region-specific Narratives","date":"2023-12-07","arxiv_id":"2312.04314","repositories_listed":1,"syntology":null},{"url":"/paper/compositional-chain-of-thought-prompting-for","slug":"compositional-chain-of-thought-prompting-for","title":"Compositional Chain-of-Thought Prompting for Large Multimodal Models","date":"2023-11-27","arxiv_id":"2311.17076","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/compositional-chain-of-thought-prompting-for#ran","syntology_url":"https://syntology.ai/paper/2311.17076","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.17076"}},"official":{"repos":["chancharikmitra/ccot"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/how-many-unicorns-are-in-this-image-a-safety","slug":"how-many-unicorns-are-in-this-image-a-safety","title":"How Many Unicorns Are in This Image? A Safety Evaluation Benchmark for Vision LLMs","date":"2023-11-27","arxiv_id":"2311.16101","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/how-many-unicorns-are-in-this-image-a-safety#ran","syntology_url":"https://syntology.ai/paper/2311.16101","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16101"}},"official":{"repos":["ucsc-vlaa/vllm-safety-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/solving-arc-visual-analogies-with-neural","slug":"solving-arc-visual-analogies-with-neural","title":"Solving ARC visual analogies with neural embeddings and vector arithmetic: A generalized method","date":"2023-11-14","arxiv_id":"2311.08083","repositories_listed":1,"syntology":null},{"url":"/paper/genome-generative-neuro-symbolic-visual","slug":"genome-generative-neuro-symbolic-visual","title":"GENOME: GenerativE Neuro-symbOlic visual reasoning by growing and reusing ModulEs","date":"2023-11-08","arxiv_id":"2311.04901","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/genome-generative-neuro-symbolic-visual#ran","syntology_url":"https://syntology.ai/paper/2311.04901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04901"}},"official":null}},{"url":"/paper/neusyre-neuro-symbolic-visual-understanding","slug":"neusyre-neuro-symbolic-visual-understanding","title":"NeuSyRE: Neuro-Symbolic Visual Understanding and Reasoning Framework based on Scene Graph Enrichment","date":"2023-11-05","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-semantic-parsing-with-2","slug":"weakly-supervised-semantic-parsing-with-2","title":"Weakly Supervised Semantic Parsing with Execution-based Spurious Program Filtering","date":"2023-11-02","arxiv_id":"2311.01161","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/weakly-supervised-semantic-parsing-with-2#ran","syntology_url":"https://syntology.ai/paper/2311.01161","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01161"}},"official":{"repos":["klee972/exec-filter"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/what-makes-for-good-visual-instructions","slug":"what-makes-for-good-visual-instructions","title":"What Makes for Good Visual Instructions? Synthesizing Complex Visual Reasoning Instructions for Visual Instruction Tuning","date":"2023-11-02","arxiv_id":"2311.01487","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/what-makes-for-good-visual-instructions#ran","syntology_url":"https://syntology.ai/paper/2311.01487","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.01487"}},"official":{"repos":["rucaibox/comvint"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/myriad-large-multimodal-model-by-applying","slug":"myriad-large-multimodal-model-by-applying","title":"Myriad: Large Multimodal Model by Applying Vision Experts for Industrial Anomaly Detection","date":"2023-10-29","arxiv_id":"2310.19070","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":2,"n_no_contract":2,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 2 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/myriad-large-multimodal-model-by-applying#ran","syntology_url":"https://syntology.ai/paper/2310.19070","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19070"}},"official":{"repos":["tzjtatata/myriad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/viclevr-a-visual-reasoning-dataset-and-hybrid","slug":"viclevr-a-visual-reasoning-dataset-and-hybrid","title":"ViCLEVR: A Visual Reasoning Dataset and Hybrid Multimodal Fusion Model for Visual Question Answering in Vietnamese","date":"2023-10-27","arxiv_id":"2310.18046","repositories_listed":1,"syntology":null},{"url":"/paper/what-s-left-concept-grounding-with-logic","slug":"what-s-left-concept-grounding-with-logic","title":"What's Left? Concept Grounding with Logic-Enhanced Foundation Models","date":"2023-10-24","arxiv_id":"2310.16035","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/what-s-left-concept-grounding-with-logic#ran","syntology_url":"https://syntology.ai/paper/2310.16035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16035"}},"official":{"repos":["joyhsu0504/left"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bongard-openworld-few-shot-reasoning-for-free","slug":"bongard-openworld-few-shot-reasoning-for-free","title":"Bongard-OpenWorld: Few-Shot Reasoning for Free-form Visual Concepts in the Real World","date":"2023-10-16","arxiv_id":"2310.10207","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bongard-openworld-few-shot-reasoning-for-free#ran","syntology_url":"https://syntology.ai/paper/2310.10207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10207"}},"official":{"repos":["joyjayng/Bongard-OpenWorld"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/interpreting-and-controlling-vision","slug":"interpreting-and-controlling-vision","title":"Interpreting and Controlling Vision Foundation Models via Text Explanations","date":"2023-10-16","arxiv_id":"2310.10591","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-differentiable-outlier-detection","slug":"implicit-differentiable-outlier-detection","title":"Implicit Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-09-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/collecting-visually-grounded-dialogue-with-a-1","slug":"collecting-visually-grounded-dialogue-with-a-1","title":"Collecting Visually-Grounded Dialogue with A Game Of Sorts","date":"2023-09-10","arxiv_id":"2309.05162","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-and-improving-chain-of-thought","slug":"measuring-and-improving-chain-of-thought","title":"Measuring and Improving Chain-of-Thought Reasoning in Vision-Language Models","date":"2023-09-08","arxiv_id":"2309.04461","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-interpretable-cross-modal","slug":"a-survey-on-interpretable-cross-modal","title":"A Survey on Interpretable Cross-modal Reasoning","date":"2023-09-05","arxiv_id":"2309.01955","repositories_listed":1,"syntology":null},{"url":"/paper/sparkles-unlocking-chats-across-multiple","slug":"sparkles-unlocking-chats-across-multiple","title":"Sparkles: Unlocking Chats Across Multiple Images for Multimodal Instruction-Following Models","date":"2023-08-31","arxiv_id":"2308.16463","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":3,"n_instrument":7,"n_unverified":3,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sparkles-unlocking-chats-across-multiple#ran","syntology_url":"https://syntology.ai/paper/2308.16463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.16463"}},"official":{"repos":["hypjudy/sparkles"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/an-examination-of-the-compositionality-of","slug":"an-examination-of-the-compositionality-of","title":"An Examination of the Compositionality of Large Generative Vision-Language Models","date":"2023-08-21","arxiv_id":"2308.10509","repositories_listed":1,"syntology":null},{"url":"/paper/vl-pet-vision-and-language-parameter","slug":"vl-pet-vision-and-language-parameter","title":"VL-PET: Vision-and-Language Parameter-Efficient Tuning via Granularity Control","date":"2023-08-18","arxiv_id":"2308.09804","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vl-pet-vision-and-language-parameter#ran","syntology_url":"https://syntology.ai/paper/2308.09804","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09804"}},"official":{"repos":["henryhzy/vl-pet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-logic-programs-by-discovering-higher","slug":"learning-logic-programs-by-discovering-higher","title":"Learning logic programs by discovering higher-order abstractions","date":"2023-08-16","arxiv_id":"2308.08334","repositories_listed":1,"syntology":null},{"url":"/paper/learning-abstract-visual-reasoning-via-task","slug":"learning-abstract-visual-reasoning-via-task","title":"Learning Abstract Visual Reasoning via Task Decomposition: A Case Study in Raven Progressive Matrices","date":"2023-08-12","arxiv_id":"2308.06528","repositories_listed":1,"syntology":null},{"url":"/paper/3d-vista-pre-trained-transformer-for-3d","slug":"3d-vista-pre-trained-transformer-for-3d","title":"3D-VisTA: Pre-trained Transformer for 3D Vision and Text Alignment","date":"2023-08-08","arxiv_id":"2308.04352","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/3d-vista-pre-trained-transformer-for-3d#ran","syntology_url":"https://syntology.ai/paper/2308.04352","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.04352"}},"official":null}},{"url":"/paper/tiny-lvlm-ehub-early-multimodal-experiments","slug":"tiny-lvlm-ehub-early-multimodal-experiments","title":"TinyLVLM-eHub: Towards Comprehensive and Efficient Evaluation for Large Vision-Language Models","date":"2023-08-07","arxiv_id":"2308.03729","repositories_listed":1,"syntology":null},{"url":"/paper/abstracting-concept-changing-rules-for","slug":"abstracting-concept-changing-rules-for","title":"Abstracting Concept-Changing Rules for Solving Raven's Progressive Matrix Problems","date":"2023-07-15","arxiv_id":"2307.07734","repositories_listed":1,"syntology":null},{"url":"/paper/learning-differentiable-logic-programs-for","slug":"learning-differentiable-logic-programs-for","title":"Learning Differentiable Logic Programs for Abstract Visual Reasoning","date":"2023-07-03","arxiv_id":"2307.00928","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/learning-differentiable-logic-programs-for#ran","syntology_url":"https://syntology.ai/paper/2307.00928","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.00928"}},"official":{"repos":["ml-research/neumann"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/stop-pre-training-adapt-visual-language","slug":"stop-pre-training-adapt-visual-language","title":"Stop Pre-Training: Adapt Visual-Language Models to Unseen Languages","date":"2023-06-29","arxiv_id":"2306.16774","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stop-pre-training-adapt-visual-language#ran","syntology_url":"https://syntology.ai/paper/2306.16774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16774"}},"official":{"repos":["yasminekaroui/clicotea"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-multimodal-large-language-models","slug":"a-survey-on-multimodal-large-language-models","title":"A Survey on Multimodal Large Language Models","date":"2023-06-23","arxiv_id":"2306.13549","repositories_listed":1,"syntology":null},{"url":"/paper/v-lol-a-diagnostic-dataset-for-visual-logical","slug":"v-lol-a-diagnostic-dataset-for-visual-logical","title":"V-LoL: A Diagnostic Dataset for Visual Logical Learning","date":"2023-06-13","arxiv_id":"2306.07743","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/v-lol-a-diagnostic-dataset-for-visual-logical#ran","syntology_url":"https://syntology.ai/paper/2306.07743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07743"}},"official":{"repos":["ml-research/vlol-dataset-gen"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/systematic-visual-reasoning-through-object-1","slug":"systematic-visual-reasoning-through-object-1","title":"Systematic Visual Reasoning through Object-Centric Relational Abstraction","date":"2023-06-04","arxiv_id":"2306.02500","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/systematic-visual-reasoning-through-object-1#ran","syntology_url":"https://syntology.ai/paper/2306.02500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02500"}},"official":{"repos":["shanka123/ocra"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visualgptscore-visio-linguistic-reasoning","slug":"visualgptscore-visio-linguistic-reasoning","title":"Revisiting the Role of Language Priors in Vision-Language Models","date":"2023-06-02","arxiv_id":"2306.01879","repositories_listed":1,"syntology":null},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/what-you-see-is-what-you-read-improving-text-1","slug":"what-you-see-is-what-you-read-improving-text-1","title":"What You See is What You Read? Improving Text-Image Alignment Evaluation","date":"2023-05-17","arxiv_id":"2305.10400","repositories_listed":1,"syntology":null},{"url":"/paper/otter-a-multi-modal-model-with-in-context","slug":"otter-a-multi-modal-model-with-in-context","title":"Otter: A Multi-Modal Model with In-Context Instruction Tuning","date":"2023-05-05","arxiv_id":"2305.03726","repositories_listed":1,"syntology":null},{"url":"/paper/visual-transformation-telling","slug":"visual-transformation-telling","title":"Visual Transformation Telling","date":"2023-05-03","arxiv_id":"2305.01928","repositories_listed":1,"syntology":null},{"url":"/paper/visual-reasoning-from-state-to-transformation","slug":"visual-reasoning-from-state-to-transformation","title":"Visual Reasoning: from State to Transformation","date":"2023-05-02","arxiv_id":"2305.01668","repositories_listed":1,"syntology":null},{"url":"/paper/going-beyond-nouns-with-vision-language","slug":"going-beyond-nouns-with-vision-language","title":"Going Beyond Nouns With Vision & Language Models Using Synthetic Data","date":"2023-03-30","arxiv_id":"2303.17590","repositories_listed":1,"syntology":null},{"url":"/paper/irfl-image-recognition-of-figurative-language","slug":"irfl-image-recognition-of-figurative-language","title":"IRFL: Image Recognition of Figurative Language","date":"2023-03-27","arxiv_id":"2303.15445","repositories_listed":1,"syntology":null},{"url":"/paper/equivariant-similarity-for-vision-language","slug":"equivariant-similarity-for-vision-language","title":"Equivariant Similarity for Vision-Language Foundation Models","date":"2023-03-25","arxiv_id":"2303.14465","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/equivariant-similarity-for-vision-language#ran","syntology_url":"https://syntology.ai/paper/2303.14465","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.14465"}},"official":{"repos":["wangt-cn/eqben"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ns3d-neuro-symbolic-grounding-of-3d-objects","slug":"ns3d-neuro-symbolic-grounding-of-3d-objects","title":"NS3D: Neuro-Symbolic Grounding of 3D Objects and Relations","date":"2023-03-23","arxiv_id":"2303.13483","repositories_listed":1,"syntology":null},{"url":"/paper/abstract-visual-reasoning-an-algebraic","slug":"abstract-visual-reasoning-an-algebraic","title":"Abstract Visual Reasoning: An Algebraic Approach for Solving Raven's Progressive Matrices","date":"2023-03-21","arxiv_id":"2303.11730","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/abstract-visual-reasoning-an-algebraic#ran","syntology_url":"https://syntology.ai/paper/2303.11730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11730"}},"official":{"repos":["xu-jingyi/algebraicmr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/is-bert-blind-exploring-the-effect-of-vision","slug":"is-bert-blind-exploring-the-effect-of-vision","title":"Is BERT Blind? Exploring the Effect of Vision-and-Language Pretraining on Visual Language Understanding","date":"2023-03-21","arxiv_id":"2303.12513","repositories_listed":1,"syntology":null},{"url":"/paper/divide-and-conquer-answering-questions-with","slug":"divide-and-conquer-answering-questions-with","title":"Divide and Conquer: Answering Questions with Object Factorization and Compositional Reasoning","date":"2023-03-18","arxiv_id":"2303.10482","repositories_listed":1,"syntology":null},{"url":"/paper/chatgpt-asks-blip-2-answers-automatic","slug":"chatgpt-asks-blip-2-answers-automatic","title":"ChatGPT Asks, BLIP-2 Answers: Automatic Questioning Towards Enriched Visual Descriptions","date":"2023-03-12","arxiv_id":"2303.06594","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chatgpt-asks-blip-2-answers-automatic#ran","syntology_url":"https://syntology.ai/paper/2303.06594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06594"}},"official":{"repos":["vision-cair/chatcaptioner"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-reason-over-visual-objects","slug":"learning-to-reason-over-visual-objects","title":"Learning to reason over visual objects","date":"2023-03-03","arxiv_id":"2303.02260","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-reason-over-visual-objects#ran","syntology_url":"https://syntology.ai/paper/2303.02260","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.02260"}},"official":{"repos":["shanka123/stsn"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/differentiable-outlier-detection-enable","slug":"differentiable-outlier-detection-enable","title":"Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-02-11","arxiv_id":"2302.05608","repositories_listed":1,"syntology":null},{"url":"/paper/multimodality-representation-learning-a","slug":"multimodality-representation-learning-a","title":"Multimodality Representation Learning: A Survey on Evolution, Pretraining and Its Applications","date":"2023-02-01","arxiv_id":"2302.00389","repositories_listed":1,"syntology":null},{"url":"/paper/see-think-confirm-interactive-prompting","slug":"see-think-confirm-interactive-prompting","title":"See, Think, Confirm: Interactive Prompting Between Vision and Language Models for Knowledge-based Visual Reasoning","date":"2023-01-12","arxiv_id":"2301.05226","repositories_listed":1,"syntology":null},{"url":"/paper/toward-building-general-foundation-models-for","slug":"toward-building-general-foundation-models-for","title":"Toward Building General Foundation Models for Language, Vision, and Vision-Language Understanding Tasks","date":"2023-01-12","arxiv_id":"2301.05065","repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-alignment-and-mutual-masking","slug":"context-aware-alignment-and-mutual-masking","title":"Context-Aware Alignment and Mutual Masking for 3D-Language Pre-Training","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/toward-multi-granularity-decision-making","slug":"toward-multi-granularity-decision-making","title":"Toward Multi-Granularity Decision-Making: Explicit Visual Reasoning with Hierarchical Knowledge","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unicode-analogies-an-anti-objectivist-visual","slug":"unicode-analogies-an-anti-objectivist-visual","title":"Unicode Analogies: An Anti-Objectivist Visual Reasoning Challenge","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-attention-congruence","slug":"cross-modal-attention-congruence","title":"Cross-modal Attention Congruence Regularization for Vision-Language Relation Alignment","date":"2022-12-20","arxiv_id":"2212.10549","repositories_listed":1,"syntology":null},{"url":"/paper/mist-multi-modal-iterative-spatial-temporal","slug":"mist-multi-modal-iterative-spatial-temporal","title":"MIST: Multi-modal Iterative Spatial-Temporal Transformer for Long-form Video Question Answering","date":"2022-12-19","arxiv_id":"2212.09522","repositories_listed":1,"syntology":null},{"url":"/paper/position-guided-text-prompt-for-vision","slug":"position-guided-text-prompt-for-vision","title":"Position-guided Text Prompt for Vision-Language Pre-training","date":"2022-12-19","arxiv_id":"2212.09737","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/position-guided-text-prompt-for-vision#ran","syntology_url":"https://syntology.ai/paper/2212.09737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09737"}},"official":{"repos":["sail-sg/ptp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/are-multimodal-models-robust-to-image-and","slug":"are-multimodal-models-robust-to-image-and","title":"Benchmarking Robustness of Multimodal Image-Text Models under Distribution Shift","date":"2022-12-15","arxiv_id":"2212.08044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/are-multimodal-models-robust-to-image-and#ran","syntology_url":"https://syntology.ai/paper/2212.08044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.08044"}},"official":null}},{"url":"/paper/vasr-visual-analogies-of-situation","slug":"vasr-visual-analogies-of-situation","title":"VASR: Visual Analogies of Situation Recognition","date":"2022-12-08","arxiv_id":"2212.04542","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vasr-visual-analogies-of-situation#ran","syntology_url":"https://syntology.ai/paper/2212.04542","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.04542"}},"official":{"repos":["vasr-dataset/vasr"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-question-answering-from-another-1","slug":"visual-question-answering-from-another-1","title":"Visual Question Answering From Another Perspective: CLEVR Mental Rotation Tests","date":"2022-12-03","arxiv_id":"2212.01639","repositories_listed":1,"syntology":null},{"url":"/paper/perceive-ground-reason-and-act-a-benchmark","slug":"perceive-ground-reason-and-act-a-benchmark","title":"Perceive, Ground, Reason, and Act: A Benchmark for General-purpose Visual Representation","date":"2022-11-28","arxiv_id":"2211.15402","repositories_listed":1,"syntology":null},{"url":"/paper/visual-programming-compositional-visual","slug":"visual-programming-compositional-visual","title":"Visual Programming: Compositional visual reasoning without training","date":"2022-11-18","arxiv_id":"2211.11559","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-programming-compositional-visual#ran","syntology_url":"https://syntology.ai/paper/2211.11559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11559"}},"official":null}},{"url":"/paper/map-modality-agnostic-uncertainty-aware","slug":"map-modality-agnostic-uncertainty-aware","title":"MAP: Multimodal Uncertainty-Aware Vision-Language Pre-training Model","date":"2022-10-11","arxiv_id":"2210.05335","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-collocate-visual-linguistic","slug":"learning-to-collocate-visual-linguistic","title":"Learning to Collocate Visual-Linguistic Neural Modules for Image Captioning","date":"2022-10-04","arxiv_id":"2210.01338","repositories_listed":1,"syntology":null},{"url":"/paper/when-and-why-vision-language-models-behave","slug":"when-and-why-vision-language-models-behave","title":"When and why vision-language models behave like bags-of-words, and what to do about it?","date":"2022-10-04","arxiv_id":"2210.01936","repositories_listed":1,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/when-and-why-vision-language-models-behave#ran","syntology_url":"https://syntology.ai/paper/2210.01936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.01936"}},"official":{"repos":["mertyg/vision-language-models-are-bows"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/a-hybrid-compositional-reasoning-approach-for","slug":"a-hybrid-compositional-reasoning-approach-for","title":"Enhancing Interpretability and Interactivity in Robot Manipulation: A Neurosymbolic Approach","date":"2022-10-03","arxiv_id":"2210.00858","repositories_listed":1,"syntology":null},{"url":"/paper/a-dual-attention-learning-network-with-word","slug":"a-dual-attention-learning-network-with-word","title":"A Dual-Attention Learning Network with Word and Sentence Embedding for Medical Visual Question Answering","date":"2022-10-01","arxiv_id":"2210.00220","repositories_listed":1,"syntology":null},{"url":"/paper/belief-revision-based-caption-re-ranker-with","slug":"belief-revision-based-caption-re-ranker-with","title":"Belief Revision based Caption Re-ranker with Visual Semantic Information","date":"2022-09-16","arxiv_id":"2209.08163","repositories_listed":1,"syntology":null},{"url":"/paper/compositional-law-parsing-with-latent-random","slug":"compositional-law-parsing-with-latent-random","title":"Compositional Law Parsing with Latent Random Functions","date":"2022-09-15","arxiv_id":"2209.09115","repositories_listed":1,"syntology":null},{"url":"/paper/viphy-probing-visible-physical-commonsense","slug":"viphy-probing-visible-physical-commonsense","title":"VIPHY: Probing \"Visible\" Physical Commonsense Knowledge","date":"2022-09-15","arxiv_id":"2209.07000","repositories_listed":1,"syntology":null},{"url":"/paper/pali-a-jointly-scaled-multilingual-language","slug":"pali-a-jointly-scaled-multilingual-language","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","date":"2022-09-14","arxiv_id":"2209.06794","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pali-a-jointly-scaled-multilingual-language#ran","syntology_url":"https://syntology.ai/paper/2209.06794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.06794"}},"official":{"repos":["google-research/big_vision"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-vision-language-pretraining-with","slug":"efficient-vision-language-pretraining-with","title":"Efficient Vision-Language Pretraining with Visual Concepts and Hierarchical Alignment","date":"2022-08-29","arxiv_id":"2208.13628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-vision-language-pretraining-with#ran","syntology_url":"https://syntology.ai/paper/2208.13628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.13628"}},"official":{"repos":["mshukor/vicha"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/winogavil-gamified-association-benchmark-to","slug":"winogavil-gamified-association-benchmark-to","title":"WinoGAViL: Gamified Association Benchmark to Challenge Vision-and-Language Models","date":"2022-07-25","arxiv_id":"2207.12576","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/winogavil-gamified-association-benchmark-to#ran","syntology_url":"https://syntology.ai/paper/2207.12576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12576"}},"official":{"repos":["winogavil/winogavil-experiments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visfis-visual-feature-importance-supervision","slug":"visfis-visual-feature-importance-supervision","title":"VisFIS: Visual Feature Importance Supervision with Right-for-the-Right-Reason Objectives","date":"2022-06-22","arxiv_id":"2206.11212","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visfis-visual-feature-importance-supervision#ran","syntology_url":"https://syntology.ai/paper/2206.11212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11212"}},"official":{"repos":["zfying/visfis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/savir-t-spatially-attentive-visual-reasoning","slug":"savir-t-spatially-attentive-visual-reasoning","title":"SAViR-T: Spatially Attentive Visual Reasoning with Transformers","date":"2022-06-18","arxiv_id":"2206.09265","repositories_listed":1,"syntology":null},{"url":"/paper/mixgen-a-new-multi-modal-data-augmentation","slug":"mixgen-a-new-multi-modal-data-augmentation","title":"MixGen: A New Multi-Modal Data Augmentation","date":"2022-06-16","arxiv_id":"2206.08358","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/mixgen-a-new-multi-modal-data-augmentation#ran","syntology_url":"https://syntology.ai/paper/2206.08358","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08358"}},"official":{"repos":["amazon-research/mix-generation"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/coarse-to-fine-vision-language-pre-training","slug":"coarse-to-fine-vision-language-pre-training","title":"Coarse-to-Fine Vision-Language Pre-training with Fusion in the Backbone","date":"2022-06-15","arxiv_id":"2206.07643","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coarse-to-fine-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2206.07643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07643"}},"official":{"repos":["microsoft/fiber"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/a-benchmark-for-compositional-visual","slug":"a-benchmark-for-compositional-visual","title":"A Benchmark for Compositional Visual Reasoning","date":"2022-06-11","arxiv_id":"2206.05379","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-benchmark-for-compositional-visual#ran","syntology_url":"https://syntology.ai/paper/2206.05379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05379"}},"official":{"repos":["aimzer/cvr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mareo-memory-and-attention-based-visual","slug":"mareo-memory-and-attention-based-visual","title":"GAMR: A Guided Attention Model for (visual) Reasoning","date":"2022-06-10","arxiv_id":"2206.04928","repositories_listed":1,"syntology":null},{"url":"/paper/expressive-scene-graph-generation-using","slug":"expressive-scene-graph-generation-using","title":"Expressive Scene Graph Generation Using Commonsense Knowledge Infusion for Visual Understanding and Reasoning","date":"2022-05-31","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cyclip-cyclic-contrastive-language-image","slug":"cyclip-cyclic-contrastive-language-image","title":"CyCLIP: Cyclic Contrastive Language-Image Pretraining","date":"2022-05-28","arxiv_id":"2205.14459","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":4,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cyclip-cyclic-contrastive-language-image#ran","syntology_url":"https://syntology.ai/paper/2205.14459","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14459"}},"official":{"repos":["goel-shashank/CyCLIP"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/bongard-hoi-benchmarking-few-shot-visual","slug":"bongard-hoi-benchmarking-few-shot-visual","title":"Bongard-HOI: Benchmarking Few-Shot Visual Reasoning for Human-Object Interactions","date":"2022-05-27","arxiv_id":"2205.13803","repositories_listed":1,"syntology":null},{"url":"/paper/multilevel-hierarchical-network-with","slug":"multilevel-hierarchical-network-with","title":"Multilevel Hierarchical Network with Multiscale Sampling for Video Question Answering","date":"2022-05-09","arxiv_id":"2205.04061","repositories_listed":1,"syntology":null},{"url":"/paper/qlevr-a-diagnostic-dataset-for","slug":"qlevr-a-diagnostic-dataset-for","title":"QLEVR: A Diagnostic Dataset for Quantificational Language and Elementary Visual Reasoning","date":"2022-05-06","arxiv_id":"2205.03075","repositories_listed":1,"syntology":null},{"url":"/paper/relvit-concept-guided-vision-transformer-for-1","slug":"relvit-concept-guided-vision-transformer-for-1","title":"RelViT: Concept-guided Vision Transformer for Visual Relational Reasoning","date":"2022-04-24","arxiv_id":"2204.11167","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":1,"n_ran_checked":3,"n_instrument":3,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/relvit-concept-guided-vision-transformer-for-1#ran","syntology_url":"https://syntology.ai/paper/2204.11167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11167"}},"official":{"repos":["NVlabs/RelViT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/clevr-x-a-visual-reasoning-dataset-for","slug":"clevr-x-a-visual-reasoning-dataset-for","title":"CLEVR-X: A Visual Reasoning Dataset for Natural Language Explanations","date":"2022-04-05","arxiv_id":"2204.02380","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/clevr-x-a-visual-reasoning-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2204.02380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02380"}},"official":{"repos":["explainableml/clevr-x"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/rex-reasoning-aware-and-grounded-explanation","slug":"rex-reasoning-aware-and-grounded-explanation","title":"REX: Reasoning-aware and Grounded Explanation","date":"2022-03-11","arxiv_id":"2203.06107","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rex-reasoning-aware-and-grounded-explanation#ran","syntology_url":"https://syntology.ai/paper/2203.06107","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.06107"}},"official":{"repos":["szzexpoi/rex"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/joint-answering-and-explanation-for-visual","slug":"joint-answering-and-explanation-for-visual","title":"Joint Answering and Explanation for Visual Commonsense Reasoning","date":"2022-02-25","arxiv_id":"2202.12626","repositories_listed":1,"syntology":null},{"url":"/paper/the-abduction-of-sherlock-holmes-a-dataset","slug":"the-abduction-of-sherlock-holmes-a-dataset","title":"The Abduction of Sherlock Holmes: A Dataset for Visual Abductive Reasoning","date":"2022-02-10","arxiv_id":"2202.04800","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-abduction-of-sherlock-holmes-a-dataset#ran","syntology_url":"https://syntology.ai/paper/2202.04800","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.04800"}},"official":null}},{"url":"/paper/clevr3d-compositional-language-and-elementary","slug":"clevr3d-compositional-language-and-elementary","title":"Comprehensive Visual Question Answering on Point Clouds through Compositional Scene Manipulation","date":"2021-12-22","arxiv_id":"2112.11691","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clevr3d-compositional-language-and-elementary#ran","syntology_url":"https://syntology.ai/paper/2112.11691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.11691"}},"official":{"repos":["yanx27/clevr3d"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/grounded-situation-recognition-with","slug":"grounded-situation-recognition-with","title":"Grounded Situation Recognition with Transformers","date":"2021-11-19","arxiv_id":"2111.10135","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/grounded-situation-recognition-with#ran","syntology_url":"https://syntology.ai/paper/2111.10135","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.10135"}},"official":{"repos":["jhcho99/gsrtr"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}}],"record_sha256":"6d496b6a8c59adbb926ba581327dd3686d932e9200aef55efc6838a722ebd148","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}