{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering-1/papers/ran/3","list_of":"/task/visual-question-answering-1","task":"Visual Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":3,"pages_in_order":4,"rows_per_page":100,"rows":[201,300],"of":378,"counts":{"archive_papers_tagged":2177,"with_a_code_link":1042,"where_syntology_ran_a_sample":378,"not_listed_spam_title":0,"listed":2177,"listed_where_code_ran":378,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":308,"every_run_a_failure_of_syntologys_instrument":70,"listed_with_a_run_with_no_instrument_failure":308,"listed_every_run_a_failure_of_syntologys_instrument":70,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering-1/papers/ran/1","prev":"/task/visual-question-answering-1/papers/ran/2","next":"/task/visual-question-answering-1/papers/ran/4","papers":[{"url":"/paper/volcano-mitigating-multimodal-hallucination","slug":"volcano-mitigating-multimodal-hallucination","title":"Volcano: Mitigating Multimodal Hallucination through Self-Feedback Guided Revision","date":"2023-11-13","arxiv_id":"2311.07362","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":3,"n_pointer_only":10,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 1 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/volcano-mitigating-multimodal-hallucination#ran","syntology_url":"https://syntology.ai/paper/2311.07362","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07362"}},"official":{"repos":["kaistai/volcano"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and","slug":"sphinx-the-joint-mixing-of-weights-tasks-and","title":"SPHINX: The Joint Mixing of Weights, Tasks, and Visual Embeddings for Multi-modal Large Language Models","date":"2023-11-13","arxiv_id":"2311.07575","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sphinx-the-joint-mixing-of-weights-tasks-and#ran","syntology_url":"https://syntology.ai/paper/2311.07575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07575"}},"official":{"repos":["alpha-vllm/llama2-accessory"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/monkey-image-resolution-and-text-label-are","slug":"monkey-image-resolution-and-text-label-are","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","date":"2023-11-11","arxiv_id":"2311.06607","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/monkey-image-resolution-and-text-label-are#ran","syntology_url":"https://syntology.ai/paper/2311.06607","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.06607"}},"official":{"repos":["yuliang-liu/monkey"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llava-plus-learning-to-use-tools-for-creating","slug":"llava-plus-learning-to-use-tools-for-creating","title":"LLaVA-Plus: Learning to Use Tools for Creating Multimodal Agents","date":"2023-11-09","arxiv_id":"2311.05437","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-plus-learning-to-use-tools-for-creating#ran","syntology_url":"https://syntology.ai/paper/2311.05437","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05437"}},"official":{"repos":["LLaVA-VL/LLaVA-Plus-Codebase"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/genome-generative-neuro-symbolic-visual","slug":"genome-generative-neuro-symbolic-visual","title":"GENOME: GenerativE Neuro-symbOlic visual reasoning by growing and reusing ModulEs","date":"2023-11-08","arxiv_id":"2311.04901","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/genome-generative-neuro-symbolic-visual#ran","syntology_url":"https://syntology.ai/paper/2311.04901","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04901"}},"official":null}},{"url":"/paper/otterhd-a-high-resolution-multi-modality","slug":"otterhd-a-high-resolution-multi-modality","title":"OtterHD: A High-Resolution Multi-modality Model","date":"2023-11-07","arxiv_id":"2311.04219","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":3,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/otterhd-a-high-resolution-multi-modality#ran","syntology_url":"https://syntology.ai/paper/2311.04219","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04219"}},"official":{"repos":["luodian/otter"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mplug-owl2-revolutionizing-multi-modal-large","slug":"mplug-owl2-revolutionizing-multi-modal-large","title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","date":"2023-11-07","arxiv_id":"2311.04257","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mplug-owl2-revolutionizing-multi-modal-large#ran","syntology_url":"https://syntology.ai/paper/2311.04257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04257"}},"official":{"repos":["x-plug/mplug-owl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/language-guided-visual-question-answering","slug":"language-guided-visual-question-answering","title":"Language Guided Visual Question Answering: Elevate Your Multimodal Language Model Using Knowledge-Enriched Prompts","date":"2023-10-31","arxiv_id":"2310.20159","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-guided-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2310.20159","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20159"}},"official":{"repos":["declare-lab/lg-vqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ehrxqa-a-multi-modal-question-answering-1","slug":"ehrxqa-a-multi-modal-question-answering-1","title":"EHRXQA: A Multi-Modal Question Answering Dataset for Electronic Health Records with Chest X-ray Images","date":"2023-10-28","arxiv_id":"2310.18652","repositories_listed":3,"syntology":{"n":20,"n_ran":20,"n_constructed":0,"n_ran_checked":17,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":0,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ehrxqa-a-multi-modal-question-answering-1#ran","syntology_url":"https://syntology.ai/paper/2310.18652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18652"}},"official":{"repos":["baeseongsu/ehrxqa","baeseongsu/mimic-cxr-vqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/3d-aware-visual-question-answering-about-1","slug":"3d-aware-visual-question-answering-about-1","title":"3D-Aware Visual Question Answering about Parts, Poses and Occlusions","date":"2023-10-27","arxiv_id":"2310.17914","repositories_listed":2,"syntology":{"n":28,"n_ran":22,"n_constructed":0,"n_ran_checked":21,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":21,"n_pointer_only":14,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 21 with no instrument failure: 0 honoured, 0 violated, 21 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/3d-aware-visual-question-answering-about-1#ran","syntology_url":"https://syntology.ai/paper/2310.17914","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17914"}},"official":{"repos":["xingruiwang/3d-aware-vqa"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/antifakeprompt-prompt-tuned-vision-language","slug":"antifakeprompt-prompt-tuned-vision-language","title":"AntifakePrompt: Prompt-Tuned Vision-Language Models are Fake Image Detectors","date":"2023-10-26","arxiv_id":"2310.17419","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/antifakeprompt-prompt-tuned-vision-language#ran","syntology_url":"https://syntology.ai/paper/2310.17419","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17419"}},"official":{"repos":["nctu-eva-lab/antifakeprompt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-simple-baseline-for-knowledge-based-visual","slug":"a-simple-baseline-for-knowledge-based-visual","title":"A Simple Baseline for Knowledge-Based Visual Question Answering","date":"2023-10-20","arxiv_id":"2310.13570","repositories_listed":0,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-simple-baseline-for-knowledge-based-visual#ran","syntology_url":"https://syntology.ai/paper/2310.13570","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13570"}},"official":null}},{"url":"/paper/frozen-transformers-in-language-models-are","slug":"frozen-transformers-in-language-models-are","title":"Frozen Transformers in Language Models Are Effective Visual Encoder Layers","date":"2023-10-19","arxiv_id":"2310.12973","repositories_listed":2,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":4,"n_honours":1,"n_violates":1,"n_no_contract":8,"n_pointer_only":8,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/frozen-transformers-in-language-models-are#ran","syntology_url":"https://syntology.ai/paper/2310.12973","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12973"}},"official":{"repos":["ziqipang/lm4visualencoding"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/minigpt-v2-large-language-model-as-a-unified","slug":"minigpt-v2-large-language-model-as-a-unified","title":"MiniGPT-v2: large language model as a unified interface for vision-language multi-task learning","date":"2023-10-14","arxiv_id":"2310.09478","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt-v2-large-language-model-as-a-unified#ran","syntology_url":"https://syntology.ai/paper/2310.09478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09478"}},"official":null}},{"url":"/paper/uncovering-hidden-connections-iterative","slug":"uncovering-hidden-connections-iterative","title":"Uncovering Hidden Connections: Iterative Search and Reasoning for Video-grounded Dialog","date":"2023-10-11","arxiv_id":"2310.07259","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/uncovering-hidden-connections-iterative#ran","syntology_url":"https://syntology.ai/paper/2310.07259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07259"}},"official":{"repos":["hyu-zhang/itr","Hyu-Zhang/ISR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rephrase-augment-reason-visual-grounding-of","slug":"rephrase-augment-reason-visual-grounding-of","title":"Rephrase, Augment, Reason: Visual Grounding of Questions for Vision-Language Models","date":"2023-10-09","arxiv_id":"2310.05861","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rephrase-augment-reason-visual-grounding-of#ran","syntology_url":"https://syntology.ai/paper/2310.05861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05861"}},"official":{"repos":["archiki/repare"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improved-baselines-with-visual-instruction","slug":"improved-baselines-with-visual-instruction","title":"Improved Baselines with Visual Instruction Tuning","date":"2023-10-05","arxiv_id":"2310.03744","repositories_listed":9,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/improved-baselines-with-visual-instruction#ran","syntology_url":"https://syntology.ai/paper/2310.03744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03744"}},"official":null}},{"url":"/paper/mathvista-evaluating-mathematical-reasoning","slug":"mathvista-evaluating-mathematical-reasoning","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","date":"2023-10-03","arxiv_id":"2310.02255","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mathvista-evaluating-mathematical-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.02255","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02255"}},"official":null}},{"url":"/paper/fine-grained-late-interaction-multi-modal-1","slug":"fine-grained-late-interaction-multi-modal-1","title":"Fine-grained Late-interaction Multi-modal Retrieval for Retrieval Augmented Visual Question Answering","date":"2023-09-29","arxiv_id":"2309.17133","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fine-grained-late-interaction-multi-modal-1#ran","syntology_url":"https://syntology.ai/paper/2309.17133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17133"}},"official":{"repos":["linweizhedragon/retrieval-augmented-visual-question-answering"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vdc-versatile-data-cleanser-for-detecting","slug":"vdc-versatile-data-cleanser-for-detecting","title":"VDC: Versatile Data Cleanser based on Visual-Linguistic Inconsistency by Multimodal Large Language Models","date":"2023-09-28","arxiv_id":"2309.16211","repositories_listed":1,"syntology":{"n":17,"n_ran":8,"n_constructed":4,"n_ran_checked":5,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"8 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/vdc-versatile-data-cleanser-for-detecting#ran","syntology_url":"https://syntology.ai/paper/2309.16211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.16211"}},"official":{"repos":["zihao-ai/vdc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/dreamllm-synergistic-multimodal-comprehension","slug":"dreamllm-synergistic-multimodal-comprehension","title":"DreamLLM: Synergistic Multimodal Comprehension and Creation","date":"2023-09-20","arxiv_id":"2309.11499","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dreamllm-synergistic-multimodal-comprehension#ran","syntology_url":"https://syntology.ai/paper/2309.11499","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.11499"}},"official":{"repos":["RunpeiDong/DreamLLM"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/an-empirical-study-of-scaling-instruct-tuned","slug":"an-empirical-study-of-scaling-instruct-tuned","title":"An Empirical Study of Scaling Instruct-Tuned Large Multimodal Models","date":"2023-09-18","arxiv_id":"2309.09958","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-of-scaling-instruct-tuned#ran","syntology_url":"https://syntology.ai/paper/2309.09958","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.09958"}},"official":{"repos":["haotian-liu/LLaVA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/instructiongpt-4-a-200-instruction-paradigm","slug":"instructiongpt-4-a-200-instruction-paradigm","title":"InstructionGPT-4: A 200-Instruction Paradigm for Fine-Tuning MiniGPT-4","date":"2023-08-23","arxiv_id":"2308.12067","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/instructiongpt-4-a-200-instruction-paradigm#ran","syntology_url":"https://syntology.ai/paper/2308.12067","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12067"}},"official":{"repos":["waltonfuture/InstructionGPT-4"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stablellava-enhanced-visual-instruction","slug":"stablellava-enhanced-visual-instruction","title":"StableLLaVA: Enhanced Visual Instruction Tuning with Synthesized Image-Dialogue Data","date":"2023-08-20","arxiv_id":"2308.10253","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stablellava-enhanced-visual-instruction#ran","syntology_url":"https://syntology.ai/paper/2308.10253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.10253"}},"official":{"repos":["icoz69/stablellava"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bliva-a-simple-multimodal-llm-for-better","slug":"bliva-a-simple-multimodal-llm-for-better","title":"BLIVA: A Simple Multimodal LLM for Better Handling of Text-Rich Visual Questions","date":"2023-08-19","arxiv_id":"2308.09936","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bliva-a-simple-multimodal-llm-for-better#ran","syntology_url":"https://syntology.ai/paper/2308.09936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09936"}},"official":{"repos":["mlpc-ucsd/bliva"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/uni-nlx-unifying-textual-explanations-for","slug":"uni-nlx-unifying-textual-explanations-for","title":"Uni-NLX: Unifying Textual Explanations for Vision and Vision-Language Tasks","date":"2023-08-17","arxiv_id":"2308.09033","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uni-nlx-unifying-textual-explanations-for#ran","syntology_url":"https://syntology.ai/paper/2308.09033","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.09033"}},"official":{"repos":["fawazsammani/uni-nlx","fawazsammani/nlxgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pro-cap-leveraging-a-frozen-vision-language","slug":"pro-cap-leveraging-a-frozen-vision-language","title":"Pro-Cap: Leveraging a Frozen Vision-Language Model for Hateful Meme Detection","date":"2023-08-16","arxiv_id":"2308.08088","repositories_listed":2,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/pro-cap-leveraging-a-frozen-vision-language#ran","syntology_url":"https://syntology.ai/paper/2308.08088","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08088"}},"official":{"repos":["social-ai-studio/pro-cap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/tech-text-guided-reconstruction-of-lifelike","slug":"tech-text-guided-reconstruction-of-lifelike","title":"TeCH: Text-guided Reconstruction of Lifelike Clothed Humans","date":"2023-08-16","arxiv_id":"2308.08545","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":10,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/tech-text-guided-reconstruction-of-lifelike#ran","syntology_url":"https://syntology.ai/paper/2308.08545","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.08545"}},"official":{"repos":["huangyangyi/tech"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/foundation-model-is-efficient-multimodal-1","slug":"foundation-model-is-efficient-multimodal-1","title":"Foundation Model is Efficient Multimodal Multitask Model Selector","date":"2023-08-11","arxiv_id":"2308.06262","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/foundation-model-is-efficient-multimodal-1#ran","syntology_url":"https://syntology.ai/paper/2308.06262","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06262"}},"official":{"repos":["opengvlab/multitask-model-selector"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn","slug":"scigraphqa-a-large-scale-synthetic-multi-turn","title":"SciGraphQA: A Large-Scale Synthetic Multi-Turn Question-Answering Dataset for Scientific Graphs","date":"2023-08-07","arxiv_id":"2308.03349","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/scigraphqa-a-large-scale-synthetic-multi-turn#ran","syntology_url":"https://syntology.ai/paper/2308.03349","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.03349"}},"official":{"repos":["findalexli/SciGraphQA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-generalist-foundation-model-for","slug":"towards-generalist-foundation-model-for","title":"Towards Generalist Foundation Model for Radiology by Leveraging Web-scale 2D&3D Medical Data","date":"2023-08-04","arxiv_id":"2308.02463","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-generalist-foundation-model-for#ran","syntology_url":"https://syntology.ai/paper/2308.02463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02463"}},"official":{"repos":["chaoyi-wu/radfm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/expert-knowledge-aware-image-difference-graph","slug":"expert-knowledge-aware-image-difference-graph","title":"Expert Knowledge-Aware Image Difference Graph Representation Learning for Difference-Aware Medical Visual Question Answering","date":"2023-07-22","arxiv_id":"2307.11986","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/expert-knowledge-aware-image-difference-graph#ran","syntology_url":"https://syntology.ai/paper/2307.11986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.11986"}},"official":{"repos":["holipori/mimic-diff-vqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/co-attention-gated-vision-language-embedding","slug":"co-attention-gated-vision-language-embedding","title":"CAT-ViL: Co-Attention Gated Vision-Language Embedding for Visual Question Localized-Answering in Robotic Surgery","date":"2023-07-11","arxiv_id":"2307.05182","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":5,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/co-attention-gated-vision-language-embedding#ran","syntology_url":"https://syntology.ai/paper/2307.05182","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05182"}},"official":{"repos":["longbai1006/cat-vil"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt4roi-instruction-tuning-large-language","slug":"gpt4roi-instruction-tuning-large-language","title":"GPT4RoI: Instruction Tuning Large Language Model on Region-of-Interest","date":"2023-07-07","arxiv_id":"2307.03601","repositories_listed":3,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/gpt4roi-instruction-tuning-large-language#ran","syntology_url":"https://syntology.ai/paper/2307.03601","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.03601"}},"official":{"repos":["jshilong/gpt4roi"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/answer-mining-from-a-pool-of-images-towards","slug":"answer-mining-from-a-pool-of-images-towards","title":"Answer Mining from a Pool of Images: Towards Retrieval-Based Visual Question Answering","date":"2023-06-29","arxiv_id":"2306.16713","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/answer-mining-from-a-pool-of-images-towards#ran","syntology_url":"https://syntology.ai/paper/2306.16713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16713"}},"official":{"repos":["Abhiram4572/mi_bart"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/shikra-unleashing-multimodal-llm-s","slug":"shikra-unleashing-multimodal-llm-s","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","date":"2023-06-27","arxiv_id":"2306.15195","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/shikra-unleashing-multimodal-llm-s#ran","syntology_url":"https://syntology.ai/paper/2306.15195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.15195"}},"official":{"repos":["shikras/shikra"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-prompting-techniques-for-zero","slug":"investigating-prompting-techniques-for-zero","title":"Investigating Prompting Techniques for Zero- and Few-Shot Visual Question Answering","date":"2023-06-16","arxiv_id":"2306.09996","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/investigating-prompting-techniques-for-zero#ran","syntology_url":"https://syntology.ai/paper/2306.09996","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.09996"}},"official":{"repos":["rabiulcste/vqazero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scalable-neural-probabilistic-answer-set","slug":"scalable-neural-probabilistic-answer-set","title":"Scalable Neural-Probabilistic Answer Set Programming","date":"2023-06-14","arxiv_id":"2306.08397","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-neural-probabilistic-answer-set#ran","syntology_url":"https://syntology.ai/paper/2306.08397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08397"}},"official":{"repos":["ml-research/slash"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crossget-cross-guided-ensemble-of-tokens-for","slug":"crossget-cross-guided-ensemble-of-tokens-for","title":"CrossGET: Cross-Guided Ensemble of Tokens for Accelerating Vision-Language Transformers","date":"2023-05-27","arxiv_id":"2305.17455","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/crossget-cross-guided-ensemble-of-tokens-for#ran","syntology_url":"https://syntology.ai/paper/2305.17455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17455"}},"official":{"repos":["sdc17/crossget"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/biomedgpt-a-unified-and-generalist-biomedical","slug":"biomedgpt-a-unified-and-generalist-biomedical","title":"BiomedGPT: A Generalist Vision-Language Foundation Model for Diverse Biomedical Tasks","date":"2023-05-26","arxiv_id":"2305.17100","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/biomedgpt-a-unified-and-generalist-biomedical#ran","syntology_url":"https://syntology.ai/paper/2305.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.17100"}},"official":{"repos":["taokz/biomedgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/nuscenes-qa-a-multi-modal-visual-question","slug":"nuscenes-qa-a-multi-modal-visual-question","title":"NuScenes-QA: A Multi-modal Visual Question Answering Benchmark for Autonomous Driving Scenario","date":"2023-05-24","arxiv_id":"2305.14836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/nuscenes-qa-a-multi-modal-visual-question#ran","syntology_url":"https://syntology.ai/paper/2305.14836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14836"}},"official":{"repos":["qiantianwen/nuscenes-qa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-art-of-socratic-questioning-zero-shot","slug":"the-art-of-socratic-questioning-zero-shot","title":"The Art of SOCRATIC QUESTIONING: Recursive Thinking with Large Language Models","date":"2023-05-24","arxiv_id":"2305.14999","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/the-art-of-socratic-questioning-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2305.14999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14999"}},"official":{"repos":["vt-nlp/socratic-questioning"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-makes-for-good-visual-tokenizers-for","slug":"what-makes-for-good-visual-tokenizers-for","title":"What Makes for Good Visual Tokenizers for Large Language Models?","date":"2023-05-20","arxiv_id":"2305.12223","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/what-makes-for-good-visual-tokenizers-for#ran","syntology_url":"https://syntology.ai/paper/2305.12223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12223"}},"official":{"repos":["tencentarc/gvt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pmc-vqa-visual-instruction-tuning-for-medical","slug":"pmc-vqa-visual-instruction-tuning-for-medical","title":"PMC-VQA: Visual Instruction Tuning for Medical Visual Question Answering","date":"2023-05-17","arxiv_id":"2305.10415","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pmc-vqa-visual-instruction-tuning-for-medical#ran","syntology_url":"https://syntology.ai/paper/2305.10415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10415"}},"official":{"repos":["xiaoman-zhang/PMC-VQA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/imad-image-augmented-multi-modal-dialogue","slug":"imad-image-augmented-multi-modal-dialogue","title":"IMAD: IMage-Augmented multi-modal Dialogue","date":"2023-05-17","arxiv_id":"2305.10512","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imad-image-augmented-multi-modal-dialogue#ran","syntology_url":"https://syntology.ai/paper/2305.10512","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10512"}},"official":{"repos":["vityavitalich/imad"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/surgicalgpt-end-to-end-language-vision-gpt","slug":"surgicalgpt-end-to-end-language-vision-gpt","title":"SurgicalGPT: End-to-End Language-Vision GPT for Visual Question Answering in Surgery","date":"2023-04-19","arxiv_id":"2304.09974","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/surgicalgpt-end-to-end-language-vision-gpt#ran","syntology_url":"https://syntology.ai/paper/2304.09974","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09974"}},"official":{"repos":["lalithjets/surgicalgpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-situation-hyper-graphs-for-video","slug":"learning-situation-hyper-graphs-for-video","title":"Learning Situation Hyper-Graphs for Video Question Answering","date":"2023-04-18","arxiv_id":"2304.08682","repositories_listed":1,"syntology":{"n":15,"n_ran":8,"n_constructed":5,"n_ran_checked":8,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":15,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/learning-situation-hyper-graphs-for-video#ran","syntology_url":"https://syntology.ai/paper/2304.08682","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08682"}},"official":{"repos":["aurooj/shg-vqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":5,"n_ran_no_instrument_failure":8,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-instruction-tuning-1","slug":"visual-instruction-tuning-1","title":"Visual Instruction Tuning","date":"2023-04-17","arxiv_id":"2304.08485","repositories_listed":13,"syntology":{"n":51,"n_ran":16,"n_constructed":6,"n_ran_checked":8,"n_instrument":8,"n_unverified":35,"n_honours":0,"n_violates":1,"n_no_contract":7,"n_pointer_only":0,"phrase":"16 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 8 where Syntology's instrument failed) · 35 unverified","sample_list":"/paper/visual-instruction-tuning-1#ran","syntology_url":"https://syntology.ai/paper/2304.08485","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08485"}},"official":{"repos":["haotian-liu/LLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":8,"ran_from_kinds":["community","listed","named_in_paper","official"]}}},{"url":"/paper/mammut-a-simple-architecture-for-joint","slug":"mammut-a-simple-architecture-for-joint","title":"MaMMUT: A Simple Architecture for Joint Learning for MultiModal Tasks","date":"2023-03-29","arxiv_id":"2303.16839","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":3,"n_no_contract":0,"n_pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 3 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mammut-a-simple-architecture-for-joint#ran","syntology_url":"https://syntology.ai/paper/2303.16839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.16839"}},"official":null}},{"url":"/paper/tifa-accurate-and-interpretable-text-to-image","slug":"tifa-accurate-and-interpretable-text-to-image","title":"TIFA: Accurate and Interpretable Text-to-Image Faithfulness Evaluation with Question Answering","date":"2023-03-21","arxiv_id":"2303.11897","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tifa-accurate-and-interpretable-text-to-image#ran","syntology_url":"https://syntology.ai/paper/2303.11897","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11897"}},"official":{"repos":["Yushi-Hu/tifa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mm-react-prompting-chatgpt-for-multimodal","slug":"mm-react-prompting-chatgpt-for-multimodal","title":"MM-REACT: Prompting ChatGPT for Multimodal Reasoning and Action","date":"2023-03-20","arxiv_id":"2303.11381","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mm-react-prompting-chatgpt-for-multimodal#ran","syntology_url":"https://syntology.ai/paper/2303.11381","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.11381"}},"official":{"repos":["microsoft/MM-REACT"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt-4-technical-report-1","slug":"gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","repositories_listed":11,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 2 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt-4-technical-report-1#ran","syntology_url":"https://syntology.ai/paper/2303.08774","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.08774"}},"official":{"repos":["openai/evals"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/open-ended-medical-visual-question-answering","slug":"open-ended-medical-visual-question-answering","title":"Open-Ended Medical Visual Question Answering Through Prefix Tuning of Language Models","date":"2023-03-10","arxiv_id":"2303.05977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-ended-medical-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2303.05977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.05977"}},"official":{"repos":["tjvsonsbeek/open-ended-medical-vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompting-large-language-models-with-answer","slug":"prompting-large-language-models-with-answer","title":"Prophet: Prompting Large Language Models with Complementary Answer Heuristics for Knowledge-based Visual Question Answering","date":"2023-03-03","arxiv_id":"2303.01903","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/prompting-large-language-models-with-answer#ran","syntology_url":"https://syntology.ai/paper/2303.01903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.01903"}},"official":{"repos":["milvlg/prophet"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-pre-trained-vision-and-language-models","slug":"can-pre-trained-vision-and-language-models","title":"Can Pre-trained Vision and Language Models Answer Visual Information-Seeking Questions?","date":"2023-02-23","arxiv_id":"2302.11713","repositories_listed":2,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-pre-trained-vision-and-language-models#ran","syntology_url":"https://syntology.ai/paper/2302.11713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.11713"}},"official":{"repos":["edchengg/infoseek_eval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mplug-2-a-modularized-multi-modal-foundation","slug":"mplug-2-a-modularized-multi-modal-foundation","title":"mPLUG-2: A Modularized Multi-modal Foundation Model Across Text, Image and Video","date":"2023-02-01","arxiv_id":"2302.00402","repositories_listed":4,"syntology":{"n":19,"n_ran":17,"n_constructed":0,"n_ran_checked":9,"n_instrument":8,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mplug-2-a-modularized-multi-modal-foundation#ran","syntology_url":"https://syntology.ai/paper/2302.00402","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00402"}},"official":{"repos":["alibaba/AliceMind"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/blip-2-bootstrapping-language-image-pre","slug":"blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","repositories_listed":17,"syntology":{"n":8,"n_ran":4,"n_constructed":4,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"4 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified; every one of the 4 samples that ran constructed an object rather than computing a result","sample_list":"/paper/blip-2-bootstrapping-language-image-pre#ran","syntology_url":"https://syntology.ai/paper/2301.12597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.12597"}},"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/super-clevr-a-virtual-benchmark-to-diagnose","slug":"super-clevr-a-virtual-benchmark-to-diagnose","title":"Super-CLEVR: A Virtual Benchmark to Diagnose Domain Robustness in Visual Reasoning","date":"2022-12-01","arxiv_id":"2212.00259","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/super-clevr-a-virtual-benchmark-to-diagnose#ran","syntology_url":"https://syntology.ai/paper/2212.00259","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.00259"}},"official":{"repos":["lizw14/super-clevr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/self-supervised-vision-language-pretraining","slug":"self-supervised-vision-language-pretraining","title":"Self-supervised vision-language pretraining for Medical visual question answering","date":"2022-11-24","arxiv_id":"2211.13594","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/self-supervised-vision-language-pretraining#ran","syntology_url":"https://syntology.ai/paper/2211.13594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.13594"}},"official":{"repos":["pengfeiliheu/m2i2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-programming-compositional-visual","slug":"visual-programming-compositional-visual","title":"Visual Programming: Compositional visual reasoning without training","date":"2022-11-18","arxiv_id":"2211.11559","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visual-programming-compositional-visual#ran","syntology_url":"https://syntology.ai/paper/2211.11559","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.11559"}},"official":null}},{"url":"/paper/i-can-t-believe-there-s-no-images-learning","slug":"i-can-t-believe-there-s-no-images-learning","title":"I Can't Believe There's No Images! Learning Visual Tasks Using only Language Supervision","date":"2022-11-17","arxiv_id":"2211.09778","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/i-can-t-believe-there-s-no-images-learning#ran","syntology_url":"https://syntology.ai/paper/2211.09778","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.09778"}},"official":{"repos":["allenai/close"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vlc-bert-visual-question-answering-with","slug":"vlc-bert-visual-question-answering-with","title":"VLC-BERT: Visual Question Answering with Contextualized Commonsense Knowledge","date":"2022-10-24","arxiv_id":"2210.13626","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlc-bert-visual-question-answering-with#ran","syntology_url":"https://syntology.ai/paper/2210.13626","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13626"}},"official":{"repos":["aditya10/vlc-bert"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/plug-and-play-vqa-zero-shot-vqa-by-conjoining","slug":"plug-and-play-vqa-zero-shot-vqa-by-conjoining","title":"Plug-and-Play VQA: Zero-shot VQA by Conjoining Large Pretrained Models with Zero Training","date":"2022-10-17","arxiv_id":"2210.08773","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/plug-and-play-vqa-zero-shot-vqa-by-conjoining#ran","syntology_url":"https://syntology.ai/paper/2210.08773","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08773"}},"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/mapl-parameter-efficient-adaptation-of","slug":"mapl-parameter-efficient-adaptation-of","title":"MAPL: Parameter-Efficient Adaptation of Unimodal Pre-Trained Models for Vision-Language Few-Shot Prompting","date":"2022-10-13","arxiv_id":"2210.07179","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mapl-parameter-efficient-adaptation-of#ran","syntology_url":"https://syntology.ai/paper/2210.07179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07179"}},"official":{"repos":["mair-lab/mapl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ernie-layout-layout-knowledge-enhanced-pre","slug":"ernie-layout-layout-knowledge-enhanced-pre","title":"ERNIE-Layout: Layout Knowledge Enhanced Pre-training for Visually-rich Document Understanding","date":"2022-10-12","arxiv_id":"2210.06155","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ernie-layout-layout-knowledge-enhanced-pre#ran","syntology_url":"https://syntology.ai/paper/2210.06155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06155"}},"official":{"repos":["PaddlePaddle/PaddleNLP"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/towards-robust-visual-question-answering","slug":"towards-robust-visual-question-answering","title":"Towards Robust Visual Question Answering: Making the Most of Biased Samples via Contrastive Learning","date":"2022-10-10","arxiv_id":"2210.04563","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-robust-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2210.04563","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04563"}},"official":{"repos":["phoebussi/mmbs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/linearly-mapping-from-image-to-text-space","slug":"linearly-mapping-from-image-to-text-space","title":"Linearly Mapping from Image to Text Space","date":"2022-09-30","arxiv_id":"2209.15162","repositories_listed":2,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/linearly-mapping-from-image-to-text-space#ran","syntology_url":"https://syntology.ai/paper/2209.15162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.15162"}},"official":{"repos":["jmerullo/limber"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tvlt-textless-vision-language-transformer","slug":"tvlt-textless-vision-language-transformer","title":"TVLT: Textless Vision-Language Transformer","date":"2022-09-28","arxiv_id":"2209.14156","repositories_listed":3,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tvlt-textless-vision-language-transformer#ran","syntology_url":"https://syntology.ai/paper/2209.14156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.14156"}},"official":{"repos":["zinengtang/tvlt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pali-a-jointly-scaled-multilingual-language","slug":"pali-a-jointly-scaled-multilingual-language","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","date":"2022-09-14","arxiv_id":"2209.06794","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pali-a-jointly-scaled-multilingual-language#ran","syntology_url":"https://syntology.ai/paper/2209.06794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.06794"}},"official":{"repos":["google-research/big_vision"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-bias-for-visual-question-answering","slug":"generative-bias-for-visual-question-answering","title":"Generative Bias for Robust Visual Question Answering","date":"2022-08-01","arxiv_id":"2208.00690","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generative-bias-for-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2208.00690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.00690"}},"official":{"repos":["chojw/genb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-causal-relational-reasoning-for","slug":"cross-modal-causal-relational-reasoning-for","title":"Cross-Modal Causal Relational Reasoning for Event-Level Visual Question Answering","date":"2022-07-26","arxiv_id":"2207.12647","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-modal-causal-relational-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2207.12647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12647"}},"official":{"repos":["hcplab-sysu/cmcir","yangliu9208/cmcir"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lako-knowledge-driven-visual-question","slug":"lako-knowledge-driven-visual-question","title":"LaKo: Knowledge-driven Visual Question Answering via Late Knowledge-to-Text Injection","date":"2022-07-26","arxiv_id":"2207.12888","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lako-knowledge-driven-visual-question#ran","syntology_url":"https://syntology.ai/paper/2207.12888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12888"}},"official":{"repos":["hackerchenzhuo/LaKo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/winogavil-gamified-association-benchmark-to","slug":"winogavil-gamified-association-benchmark-to","title":"WinoGAViL: Gamified Association Benchmark to Challenge Vision-and-Language Models","date":"2022-07-25","arxiv_id":"2207.12576","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/winogavil-gamified-association-benchmark-to#ran","syntology_url":"https://syntology.ai/paper/2207.12576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12576"}},"official":{"repos":["winogavil/winogavil-experiments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/surgical-vqa-visual-question-answering-in","slug":"surgical-vqa-visual-question-answering-in","title":"Surgical-VQA: Visual Question Answering in Surgical Scenes using Transformer","date":"2022-06-22","arxiv_id":"2206.11053","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/surgical-vqa-visual-question-answering-in#ran","syntology_url":"https://syntology.ai/paper/2206.11053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11053"}},"official":{"repos":["lalithjets/surgical_vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visfis-visual-feature-importance-supervision","slug":"visfis-visual-feature-importance-supervision","title":"VisFIS: Visual Feature Importance Supervision with Right-for-the-Right-Reason Objectives","date":"2022-06-22","arxiv_id":"2206.11212","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visfis-visual-feature-importance-supervision#ran","syntology_url":"https://syntology.ai/paper/2206.11212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11212"}},"official":{"repos":["zfying/visfis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-video-question-answering-via-frozen","slug":"zero-shot-video-question-answering-via-frozen","title":"Zero-Shot Video Question Answering via Frozen Bidirectional Language Models","date":"2022-06-16","arxiv_id":"2206.08155","repositories_listed":3,"syntology":{"n":34,"n_ran":14,"n_constructed":9,"n_ran_checked":14,"n_instrument":0,"n_unverified":20,"n_honours":1,"n_violates":1,"n_no_contract":12,"n_pointer_only":1,"phrase":"14 ran (of which 9 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 20 unverified","sample_list":"/paper/zero-shot-video-question-answering-via-frozen#ran","syntology_url":"https://syntology.ai/paper/2206.08155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08155"}},"official":{"repos":["antoyang/FrozenBiLM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["listed"]}}},{"url":"/paper/coarse-to-fine-vision-language-pre-training","slug":"coarse-to-fine-vision-language-pre-training","title":"Coarse-to-Fine Vision-Language Pre-training with Fusion in the Backbone","date":"2022-06-15","arxiv_id":"2206.07643","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coarse-to-fine-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2206.07643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07643"}},"official":{"repos":["microsoft/fiber"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/a-okvqa-a-benchmark-for-visual-question","slug":"a-okvqa-a-benchmark-for-visual-question","title":"A-OKVQA: A Benchmark for Visual Question Answering using World Knowledge","date":"2022-06-03","arxiv_id":"2206.01718","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-okvqa-a-benchmark-for-visual-question#ran","syntology_url":"https://syntology.ai/paper/2206.01718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01718"}},"official":{"repos":["allenai/aokvqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revive-regional-visual-representation-matters","slug":"revive-regional-visual-representation-matters","title":"REVIVE: Regional Visual Representation Matters in Knowledge-Based Visual Question Answering","date":"2022-06-02","arxiv_id":"2206.01201","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/revive-regional-visual-representation-matters#ran","syntology_url":"https://syntology.ai/paper/2206.01201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01201"}},"official":{"repos":["yzleroy/revive"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/coca-contrastive-captioners-are-image-text","slug":"coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","arxiv_id":"2205.01917","repositories_listed":6,"syntology":{"n":17,"n_ran":10,"n_constructed":5,"n_ran_checked":10,"n_instrument":0,"n_unverified":7,"n_honours":2,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/coca-contrastive-captioners-are-image-text#ran","syntology_url":"https://syntology.ai/paper/2205.01917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01917"}},"official":null}},{"url":"/paper/flamingo-a-visual-language-model-for-few-shot-1","slug":"flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","arxiv_id":"2204.14198","repositories_listed":5,"syntology":{"n":24,"n_ran":18,"n_constructed":6,"n_ran_checked":12,"n_instrument":6,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":8,"phrase":"18 ran (of which 6 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/flamingo-a-visual-language-model-for-few-shot-1#ran","syntology_url":"https://syntology.ai/paper/2204.14198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.14198"}},"official":null}},{"url":"/paper/reliable-visual-question-answering-abstain","slug":"reliable-visual-question-answering-abstain","title":"Reliable Visual Question Answering: Abstain Rather Than Answer Incorrectly","date":"2022-04-28","arxiv_id":"2204.13631","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reliable-visual-question-answering-abstain#ran","syntology_url":"https://syntology.ai/paper/2204.13631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13631"}},"official":{"repos":["facebookresearch/reliable_vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grit-general-robust-image-task-benchmark","slug":"grit-general-robust-image-task-benchmark","title":"GRIT: General Robust Image Task Benchmark","date":"2022-04-28","arxiv_id":"2204.13653","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grit-general-robust-image-task-benchmark#ran","syntology_url":"https://syntology.ai/paper/2204.13653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13653"}},"official":{"repos":["allenai/grit_official"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/clevr-x-a-visual-reasoning-dataset-for","slug":"clevr-x-a-visual-reasoning-dataset-for","title":"CLEVR-X: A Visual Reasoning Dataset for Natural Language Explanations","date":"2022-04-05","arxiv_id":"2204.02380","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/clevr-x-a-visual-reasoning-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2204.02380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02380"}},"official":{"repos":["explainableml/clevr-x"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-interpret-an-interactive-visualization","slug":"vl-interpret-an-interactive-visualization","title":"VL-InterpreT: An Interactive Visualization Tool for Interpreting Vision-Language Transformers","date":"2022-03-30","arxiv_id":"2203.17247","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vl-interpret-an-interactive-visualization#ran","syntology_url":"https://syntology.ai/paper/2203.17247","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.17247"}},"official":{"repos":["intellabs/vl-interpret"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-stitch-in-time-saves-nine-a-train-time","slug":"a-stitch-in-time-saves-nine-a-train-time","title":"A Stitch in Time Saves Nine: A Train-Time Regularizing Loss for Improved Neural Network Calibration","date":"2022-03-25","arxiv_id":"2203.13834","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-stitch-in-time-saves-nine-a-train-time#ran","syntology_url":"https://syntology.ai/paper/2203.13834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13834"}},"official":{"repos":["mdca-loss/mdca-calibration"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/mukea-multimodal-knowledge-extraction-and","slug":"mukea-multimodal-knowledge-extraction-and","title":"MuKEA: Multimodal Knowledge Extraction and Accumulation for Knowledge-based Visual Question Answering","date":"2022-03-17","arxiv_id":"2203.09138","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mukea-multimodal-knowledge-extraction-and#ran","syntology_url":"https://syntology.ai/paper/2203.09138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09138"}},"official":{"repos":["andersonstra/mukea"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vision-language-pre-training-with-triple","slug":"vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","arxiv_id":"2202.10401","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vision-language-pre-training-with-triple#ran","syntology_url":"https://syntology.ai/paper/2202.10401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10401"}},"official":{"repos":["uta-smile/TCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-architectures-tasks-and-modalities","slug":"unifying-architectures-tasks-and-modalities","title":"OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework","date":"2022-02-07","arxiv_id":"2202.03052","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-architectures-tasks-and-modalities#ran","syntology_url":"https://syntology.ai/paper/2202.03052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03052"}},"official":{"repos":["ofa-sys/ofa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grounding-answers-for-visual-questions-asked","slug":"grounding-answers-for-visual-questions-asked","title":"Grounding Answers for Visual Questions Asked by Visually Impaired People","date":"2022-02-04","arxiv_id":"2202.01993","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounding-answers-for-visual-questions-asked#ran","syntology_url":"https://syntology.ai/paper/2202.01993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.01993"}},"official":{"repos":["ccychongyanchen/vizwizvqagroundingcrowdsourcing"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/compositionality-as-lexical-symmetry","slug":"compositionality-as-lexical-symmetry","title":"Compositionality as Lexical Symmetry","date":"2022-01-30","arxiv_id":"2201.12926","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/compositionality-as-lexical-symmetry#ran","syntology_url":"https://syntology.ai/paper/2201.12926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12926"}},"official":{"repos":["ekinakyurek/lexsym"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/iglue-a-benchmark-for-transfer-learning","slug":"iglue-a-benchmark-for-transfer-learning","title":"IGLUE: A Benchmark for Transfer Learning across Modalities, Tasks, and Languages","date":"2022-01-27","arxiv_id":"2201.11732","repositories_listed":3,"syntology":{"n":21,"n_ran":17,"n_constructed":2,"n_ran_checked":11,"n_instrument":6,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":9,"n_pointer_only":3,"phrase":"17 ran (of which 2 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 0 violated, 9 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/iglue-a-benchmark-for-transfer-learning#ran","syntology_url":"https://syntology.ai/paper/2201.11732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.11732"}},"official":{"repos":["e-bug/iglue","e-bug/volta"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":2,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/latr-layout-aware-transformer-for-scene-text","slug":"latr-layout-aware-transformer-for-scene-text","title":"LaTr: Layout-Aware Transformer for Scene-Text VQA","date":"2021-12-23","arxiv_id":"2112.12494","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/latr-layout-aware-transformer-for-scene-text#ran","syntology_url":"https://syntology.ai/paper/2112.12494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.12494"}},"official":null}},{"url":"/paper/clevr3d-compositional-language-and-elementary","slug":"clevr3d-compositional-language-and-elementary","title":"Comprehensive Visual Question Answering on Point Clouds through Compositional Scene Manipulation","date":"2021-12-22","arxiv_id":"2112.11691","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clevr3d-compositional-language-and-elementary#ran","syntology_url":"https://syntology.ai/paper/2112.11691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.11691"}},"official":{"repos":["yanx27/clevr3d"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/crossing-the-format-boundary-of-text-and","slug":"crossing-the-format-boundary-of-text-and","title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","date":"2021-11-23","arxiv_id":"2111.12085","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crossing-the-format-boundary-of-text-and#ran","syntology_url":"https://syntology.ai/paper/2111.12085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12085"}},"official":{"repos":["microsoft/UniTAB"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/perceptual-score-what-data-modalities-does","slug":"perceptual-score-what-data-modalities-does","title":"Perceptual Score: What Data Modalities Does Your Model Perceive?","date":"2021-10-27","arxiv_id":"2110.14375","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/perceptual-score-what-data-modalities-does#ran","syntology_url":"https://syntology.ai/paper/2110.14375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14375"}},"official":{"repos":["itaigat/perceptual-score"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alignment-attention-by-matching-key-and-query","slug":"alignment-attention-by-matching-key-and-query","title":"Alignment Attention by Matching Key and Query Distributions","date":"2021-10-25","arxiv_id":"2110.12567","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/alignment-attention-by-matching-key-and-query#ran","syntology_url":"https://syntology.ai/paper/2110.12567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.12567"}},"official":{"repos":["szhang42/alignment_attention"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/iconqa-a-new-benchmark-for-abstract-diagram","slug":"iconqa-a-new-benchmark-for-abstract-diagram","title":"IconQA: A New Benchmark for Abstract Diagram Understanding and Visual Language Reasoning","date":"2021-10-25","arxiv_id":"2110.13214","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/iconqa-a-new-benchmark-for-abstract-diagram#ran","syntology_url":"https://syntology.ai/paper/2110.13214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13214"}},"official":{"repos":["lupantech/iconqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/label-descriptive-patterns-and-their","slug":"label-descriptive-patterns-and-their","title":"Label-Descriptive Patterns and Their Application to Characterizing Classification Errors","date":"2021-10-18","arxiv_id":"2110.09599","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/label-descriptive-patterns-and-their#ran","syntology_url":"https://syntology.ai/paper/2110.09599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.09599"}},"official":{"repos":["uds-lsv/premise"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/coarse-to-fine-reasoning-for-visual-question","slug":"coarse-to-fine-reasoning-for-visual-question","title":"Coarse-to-Fine Reasoning for Visual Question Answering","date":"2021-10-06","arxiv_id":"2110.02526","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":3,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/coarse-to-fine-reasoning-for-visual-question#ran","syntology_url":"https://syntology.ai/paper/2110.02526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.02526"}},"official":{"repos":["aioz-ai/cfr_vqa","aioz-ai/crf_vqa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}}],"record_sha256":"4ed18cc140ea1065ae2a6a321f5515444ccb1130b91233c057ce86b64f41835e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}