{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/ran/3","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":3,"pages_in_order":4,"rows_per_page":100,"rows":[201,300],"of":359,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering/papers/ran/1","prev":"/task/visual-question-answering/papers/ran/2","next":"/task/visual-question-answering/papers/ran/4","papers":[{"url":"/paper/mapl-parameter-efficient-adaptation-of","slug":"mapl-parameter-efficient-adaptation-of","title":"MAPL: Parameter-Efficient Adaptation of Unimodal Pre-Trained Models for Vision-Language Few-Shot Prompting","date":"2022-10-13","arxiv_id":"2210.07179","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mapl-parameter-efficient-adaptation-of#ran","syntology_url":"https://syntology.ai/paper/2210.07179","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.07179"}},"official":{"repos":["mair-lab/mapl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ernie-layout-layout-knowledge-enhanced-pre","slug":"ernie-layout-layout-knowledge-enhanced-pre","title":"ERNIE-Layout: Layout Knowledge Enhanced Pre-training for Visually-rich Document Understanding","date":"2022-10-12","arxiv_id":"2210.06155","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ernie-layout-layout-knowledge-enhanced-pre#ran","syntology_url":"https://syntology.ai/paper/2210.06155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.06155"}},"official":{"repos":["PaddlePaddle/PaddleNLP"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/towards-robust-visual-question-answering","slug":"towards-robust-visual-question-answering","title":"Towards Robust Visual Question Answering: Making the Most of Biased Samples via Contrastive Learning","date":"2022-10-10","arxiv_id":"2210.04563","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-robust-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2210.04563","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.04563"}},"official":{"repos":["phoebussi/mmbs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/pix2struct-screenshot-parsing-as-pretraining","slug":"pix2struct-screenshot-parsing-as-pretraining","title":"Pix2Struct: Screenshot Parsing as Pretraining for Visual Language Understanding","date":"2022-10-07","arxiv_id":"2210.03347","repositories_listed":4,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pix2struct-screenshot-parsing-as-pretraining#ran","syntology_url":"https://syntology.ai/paper/2210.03347","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.03347"}},"official":{"repos":["google-research/pix2struct"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/linearly-mapping-from-image-to-text-space","slug":"linearly-mapping-from-image-to-text-space","title":"Linearly Mapping from Image to Text Space","date":"2022-09-30","arxiv_id":"2209.15162","repositories_listed":2,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":10,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/linearly-mapping-from-image-to-text-space#ran","syntology_url":"https://syntology.ai/paper/2209.15162","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.15162"}},"official":{"repos":["jmerullo/limber"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tvlt-textless-vision-language-transformer","slug":"tvlt-textless-vision-language-transformer","title":"TVLT: Textless Vision-Language Transformer","date":"2022-09-28","arxiv_id":"2209.14156","repositories_listed":3,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tvlt-textless-vision-language-transformer#ran","syntology_url":"https://syntology.ai/paper/2209.14156","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.14156"}},"official":{"repos":["zinengtang/tvlt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learn-to-explain-multimodal-reasoning-via","slug":"learn-to-explain-multimodal-reasoning-via","title":"Learn to Explain: Multimodal Reasoning via Thought Chains for Science Question Answering","date":"2022-09-20","arxiv_id":"2209.09513","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learn-to-explain-multimodal-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2209.09513","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.09513"}},"official":{"repos":["lupantech/ScienceQA"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/panoramic-vision-transformer-for-saliency","slug":"panoramic-vision-transformer-for-saliency","title":"Panoramic Vision Transformer for Saliency Detection in 360° Videos","date":"2022-09-19","arxiv_id":"2209.08956","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/panoramic-vision-transformer-for-saliency#ran","syntology_url":"https://syntology.ai/paper/2209.08956","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.08956"}},"official":{"repos":["hs-yn/paver"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pali-a-jointly-scaled-multilingual-language","slug":"pali-a-jointly-scaled-multilingual-language","title":"PaLI: A Jointly-Scaled Multilingual Language-Image Model","date":"2022-09-14","arxiv_id":"2209.06794","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":1,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pali-a-jointly-scaled-multilingual-language#ran","syntology_url":"https://syntology.ai/paper/2209.06794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.06794"}},"official":{"repos":["google-research/big_vision"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-vision-language-pretraining-with","slug":"efficient-vision-language-pretraining-with","title":"Efficient Vision-Language Pretraining with Visual Concepts and Hierarchical Alignment","date":"2022-08-29","arxiv_id":"2208.13628","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-vision-language-pretraining-with#ran","syntology_url":"https://syntology.ai/paper/2208.13628","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.13628"}},"official":{"repos":["mshukor/vicha"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompt-tuning-for-generative-multimodal","slug":"prompt-tuning-for-generative-multimodal","title":"Prompt Tuning for Generative Multimodal Pretrained Models","date":"2022-08-04","arxiv_id":"2208.02532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompt-tuning-for-generative-multimodal#ran","syntology_url":"https://syntology.ai/paper/2208.02532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.02532"}},"official":{"repos":["ofa-sys/ofa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generative-bias-for-visual-question-answering","slug":"generative-bias-for-visual-question-answering","title":"Generative Bias for Robust Visual Question Answering","date":"2022-08-01","arxiv_id":"2208.00690","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generative-bias-for-visual-question-answering#ran","syntology_url":"https://syntology.ai/paper/2208.00690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2208.00690"}},"official":{"repos":["chojw/genb"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-modal-causal-relational-reasoning-for","slug":"cross-modal-causal-relational-reasoning-for","title":"Cross-Modal Causal Relational Reasoning for Event-Level Visual Question Answering","date":"2022-07-26","arxiv_id":"2207.12647","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cross-modal-causal-relational-reasoning-for#ran","syntology_url":"https://syntology.ai/paper/2207.12647","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12647"}},"official":{"repos":["hcplab-sysu/cmcir","yangliu9208/cmcir"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lako-knowledge-driven-visual-question","slug":"lako-knowledge-driven-visual-question","title":"LaKo: Knowledge-driven Visual Question Answering via Late Knowledge-to-Text Injection","date":"2022-07-26","arxiv_id":"2207.12888","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lako-knowledge-driven-visual-question#ran","syntology_url":"https://syntology.ai/paper/2207.12888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12888"}},"official":{"repos":["hackerchenzhuo/LaKo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/winogavil-gamified-association-benchmark-to","slug":"winogavil-gamified-association-benchmark-to","title":"WinoGAViL: Gamified Association Benchmark to Challenge Vision-and-Language Models","date":"2022-07-25","arxiv_id":"2207.12576","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/winogavil-gamified-association-benchmark-to#ran","syntology_url":"https://syntology.ai/paper/2207.12576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.12576"}},"official":{"repos":["winogavil/winogavil-experiments"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/video-graph-transformer-for-video-question","slug":"video-graph-transformer-for-video-question","title":"Video Graph Transformer for Video Question Answering","date":"2022-07-12","arxiv_id":"2207.05342","repositories_listed":1,"syntology":{"n":14,"n_ran":9,"n_constructed":5,"n_ran_checked":8,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/video-graph-transformer-for-video-question#ran","syntology_url":"https://syntology.ai/paper/2207.05342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.05342"}},"official":{"repos":["sail-sg/vgt"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":5,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/surgical-vqa-visual-question-answering-in","slug":"surgical-vqa-visual-question-answering-in","title":"Surgical-VQA: Visual Question Answering in Surgical Scenes using Transformer","date":"2022-06-22","arxiv_id":"2206.11053","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/surgical-vqa-visual-question-answering-in#ran","syntology_url":"https://syntology.ai/paper/2206.11053","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11053"}},"official":{"repos":["lalithjets/surgical_vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visfis-visual-feature-importance-supervision","slug":"visfis-visual-feature-importance-supervision","title":"VisFIS: Visual Feature Importance Supervision with Right-for-the-Right-Reason Objectives","date":"2022-06-22","arxiv_id":"2206.11212","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visfis-visual-feature-importance-supervision#ran","syntology_url":"https://syntology.ai/paper/2206.11212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.11212"}},"official":{"repos":["zfying/visfis"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-video-question-answering-via-frozen","slug":"zero-shot-video-question-answering-via-frozen","title":"Zero-Shot Video Question Answering via Frozen Bidirectional Language Models","date":"2022-06-16","arxiv_id":"2206.08155","repositories_listed":3,"syntology":{"n":34,"n_ran":14,"n_constructed":9,"n_ran_checked":14,"n_instrument":0,"n_unverified":20,"n_honours":1,"n_violates":1,"n_no_contract":12,"n_pointer_only":1,"phrase":"14 ran (of which 9 constructed an object rather than computing a result; 14 with no instrument failure: 1 honoured, 1 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 20 unverified","sample_list":"/paper/zero-shot-video-question-answering-via-frozen#ran","syntology_url":"https://syntology.ai/paper/2206.08155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08155"}},"official":{"repos":["antoyang/FrozenBiLM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["listed"]}}},{"url":"/paper/coarse-to-fine-vision-language-pre-training","slug":"coarse-to-fine-vision-language-pre-training","title":"Coarse-to-Fine Vision-Language Pre-training with Fusion in the Backbone","date":"2022-06-15","arxiv_id":"2206.07643","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/coarse-to-fine-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2206.07643","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07643"}},"official":{"repos":["microsoft/fiber"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/lst-ladder-side-tuning-for-parameter-and","slug":"lst-ladder-side-tuning-for-parameter-and","title":"LST: Ladder Side-Tuning for Parameter and Memory Efficient Transfer Learning","date":"2022-06-13","arxiv_id":"2206.06522","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/lst-ladder-side-tuning-for-parameter-and#ran","syntology_url":"https://syntology.ai/paper/2206.06522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.06522"}},"official":{"repos":["ylsung/ladder-side-tuning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/glipv2-unifying-localization-and-vision","slug":"glipv2-unifying-localization-and-vision","title":"GLIPv2: Unifying Localization and Vision-Language Understanding","date":"2022-06-12","arxiv_id":"2206.05836","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glipv2-unifying-localization-and-vision#ran","syntology_url":"https://syntology.ai/paper/2206.05836","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05836"}},"official":{"repos":["microsoft/GLIP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-okvqa-a-benchmark-for-visual-question","slug":"a-okvqa-a-benchmark-for-visual-question","title":"A-OKVQA: A Benchmark for Visual Question Answering using World Knowledge","date":"2022-06-03","arxiv_id":"2206.01718","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-okvqa-a-benchmark-for-visual-question#ran","syntology_url":"https://syntology.ai/paper/2206.01718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01718"}},"official":{"repos":["allenai/aokvqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revive-regional-visual-representation-matters","slug":"revive-regional-visual-representation-matters","title":"REVIVE: Regional Visual Representation Matters in Knowledge-Based Visual Question Answering","date":"2022-06-02","arxiv_id":"2206.01201","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/revive-regional-visual-representation-matters#ran","syntology_url":"https://syntology.ai/paper/2206.01201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01201"}},"official":{"repos":["yzleroy/revive"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/git-a-generative-image-to-text-transformer","slug":"git-a-generative-image-to-text-transformer","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","date":"2022-05-27","arxiv_id":"2205.14100","repositories_listed":1,"syntology":{"n":21,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/git-a-generative-image-to-text-transformer#ran","syntology_url":"https://syntology.ai/paper/2205.14100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14100"}},"official":{"repos":["microsoft/GenerativeImage2Text"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/pevl-position-enhanced-pre-training-and","slug":"pevl-position-enhanced-pre-training-and","title":"PEVL: Position-enhanced Pre-training and Prompt Tuning for Vision-language Models","date":"2022-05-23","arxiv_id":"2205.11169","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pevl-position-enhanced-pre-training-and#ran","syntology_url":"https://syntology.ai/paper/2205.11169","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.11169"}},"official":{"repos":["thunlp/pevl"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/coca-contrastive-captioners-are-image-text","slug":"coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","arxiv_id":"2205.01917","repositories_listed":6,"syntology":{"n":17,"n_ran":10,"n_constructed":5,"n_ran_checked":10,"n_instrument":0,"n_unverified":7,"n_honours":2,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 10 with no instrument failure: 2 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/coca-contrastive-captioners-are-image-text#ran","syntology_url":"https://syntology.ai/paper/2205.01917","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.01917"}},"official":null}},{"url":"/paper/flamingo-a-visual-language-model-for-few-shot-1","slug":"flamingo-a-visual-language-model-for-few-shot-1","title":"Flamingo: a Visual Language Model for Few-Shot Learning","date":"2022-04-29","arxiv_id":"2204.14198","repositories_listed":5,"syntology":{"n":24,"n_ran":18,"n_constructed":6,"n_ran_checked":12,"n_instrument":6,"n_unverified":6,"n_honours":0,"n_violates":1,"n_no_contract":11,"n_pointer_only":8,"phrase":"18 ran (of which 6 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 6 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/flamingo-a-visual-language-model-for-few-shot-1#ran","syntology_url":"https://syntology.ai/paper/2204.14198","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.14198"}},"official":null}},{"url":"/paper/reliable-visual-question-answering-abstain","slug":"reliable-visual-question-answering-abstain","title":"Reliable Visual Question Answering: Abstain Rather Than Answer Incorrectly","date":"2022-04-28","arxiv_id":"2204.13631","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reliable-visual-question-answering-abstain#ran","syntology_url":"https://syntology.ai/paper/2204.13631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13631"}},"official":{"repos":["facebookresearch/reliable_vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grit-general-robust-image-task-benchmark","slug":"grit-general-robust-image-task-benchmark","title":"GRIT: General Robust Image Task Benchmark","date":"2022-04-28","arxiv_id":"2204.13653","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grit-general-robust-image-task-benchmark#ran","syntology_url":"https://syntology.ai/paper/2204.13653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13653"}},"official":{"repos":["allenai/grit_official"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/relvit-concept-guided-vision-transformer-for-1","slug":"relvit-concept-guided-vision-transformer-for-1","title":"RelViT: Concept-guided Vision Transformer for Visual Relational Reasoning","date":"2022-04-24","arxiv_id":"2204.11167","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":1,"n_ran_checked":3,"n_instrument":3,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/relvit-concept-guided-vision-transformer-for-1#ran","syntology_url":"https://syntology.ai/paper/2204.11167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11167"}},"official":{"repos":["NVlabs/RelViT"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/clevr-x-a-visual-reasoning-dataset-for","slug":"clevr-x-a-visual-reasoning-dataset-for","title":"CLEVR-X: A Visual Reasoning Dataset for Natural Language Explanations","date":"2022-04-05","arxiv_id":"2204.02380","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/clevr-x-a-visual-reasoning-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2204.02380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02380"}},"official":{"repos":["explainableml/clevr-x"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-document-recognition-and","slug":"end-to-end-document-recognition-and","title":"End-to-end Document Recognition and Understanding with Dessurt","date":"2022-03-30","arxiv_id":"2203.16618","repositories_listed":2,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/end-to-end-document-recognition-and#ran","syntology_url":"https://syntology.ai/paper/2203.16618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16618"}},"official":{"repos":["herobd/dessurt"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vl-interpret-an-interactive-visualization","slug":"vl-interpret-an-interactive-visualization","title":"VL-InterpreT: An Interactive Visualization Tool for Interpreting Vision-Language Transformers","date":"2022-03-30","arxiv_id":"2203.17247","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vl-interpret-an-interactive-visualization#ran","syntology_url":"https://syntology.ai/paper/2203.17247","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.17247"}},"official":{"repos":["intellabs/vl-interpret"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mukea-multimodal-knowledge-extraction-and","slug":"mukea-multimodal-knowledge-extraction-and","title":"MuKEA: Multimodal Knowledge Extraction and Accumulation for Knowledge-based Visual Question Answering","date":"2022-03-17","arxiv_id":"2203.09138","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mukea-multimodal-knowledge-extraction-and#ran","syntology_url":"https://syntology.ai/paper/2203.09138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09138"}},"official":{"repos":["andersonstra/mukea"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/assistq-affordance-centric-question-driven","slug":"assistq-affordance-centric-question-driven","title":"AssistQ: Affordance-centric Question-driven Task Completion for Egocentric Assistant","date":"2022-03-08","arxiv_id":"2203.04203","repositories_listed":4,"syntology":{"n":7,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/assistq-affordance-centric-question-driven#ran","syntology_url":"https://syntology.ai/paper/2203.04203","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.04203"}},"official":{"repos":["showlab/Q2A"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/vision-language-pre-training-with-triple","slug":"vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","arxiv_id":"2202.10401","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vision-language-pre-training-with-triple#ran","syntology_url":"https://syntology.ai/paper/2202.10401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10401"}},"official":{"repos":["uta-smile/TCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unifying-architectures-tasks-and-modalities","slug":"unifying-architectures-tasks-and-modalities","title":"OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework","date":"2022-02-07","arxiv_id":"2202.03052","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-architectures-tasks-and-modalities#ran","syntology_url":"https://syntology.ai/paper/2202.03052","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.03052"}},"official":{"repos":["ofa-sys/ofa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grounding-answers-for-visual-questions-asked","slug":"grounding-answers-for-visual-questions-asked","title":"Grounding Answers for Visual Questions Asked by Visually Impaired People","date":"2022-02-04","arxiv_id":"2202.01993","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounding-answers-for-visual-questions-asked#ran","syntology_url":"https://syntology.ai/paper/2202.01993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.01993"}},"official":{"repos":["ccychongyanchen/vizwizvqagroundingcrowdsourcing"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/compositionality-as-lexical-symmetry","slug":"compositionality-as-lexical-symmetry","title":"Compositionality as Lexical Symmetry","date":"2022-01-30","arxiv_id":"2201.12926","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/compositionality-as-lexical-symmetry#ran","syntology_url":"https://syntology.ai/paper/2201.12926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12926"}},"official":{"repos":["ekinakyurek/lexsym"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/iglue-a-benchmark-for-transfer-learning","slug":"iglue-a-benchmark-for-transfer-learning","title":"IGLUE: A Benchmark for Transfer Learning across Modalities, Tasks, and Languages","date":"2022-01-27","arxiv_id":"2201.11732","repositories_listed":3,"syntology":{"n":21,"n_ran":17,"n_constructed":2,"n_ran_checked":11,"n_instrument":6,"n_unverified":4,"n_honours":2,"n_violates":0,"n_no_contract":9,"n_pointer_only":3,"phrase":"17 ran (of which 2 constructed an object rather than computing a result; 11 with no instrument failure: 2 honoured, 0 violated, 9 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/iglue-a-benchmark-for-transfer-learning#ran","syntology_url":"https://syntology.ai/paper/2201.11732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.11732"}},"official":{"repos":["e-bug/iglue","e-bug/volta"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":2,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/latr-layout-aware-transformer-for-scene-text","slug":"latr-layout-aware-transformer-for-scene-text","title":"LaTr: Layout-Aware Transformer for Scene-Text VQA","date":"2021-12-23","arxiv_id":"2112.12494","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/latr-layout-aware-transformer-for-scene-text#ran","syntology_url":"https://syntology.ai/paper/2112.12494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.12494"}},"official":null}},{"url":"/paper/clevr3d-compositional-language-and-elementary","slug":"clevr3d-compositional-language-and-elementary","title":"Comprehensive Visual Question Answering on Point Clouds through Compositional Scene Manipulation","date":"2021-12-22","arxiv_id":"2112.11691","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clevr3d-compositional-language-and-elementary#ran","syntology_url":"https://syntology.ai/paper/2112.11691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.11691"}},"official":{"repos":["yanx27/clevr3d"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/video-as-conditional-graph-hierarchy-for","slug":"video-as-conditional-graph-hierarchy-for","title":"Video as Conditional Graph Hierarchy for Multi-Granular Question Answering","date":"2021-12-12","arxiv_id":"2112.06197","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-as-conditional-graph-hierarchy-for#ran","syntology_url":"https://syntology.ai/paper/2112.06197","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.06197"}},"official":{"repos":["doc-doc/hqga"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/donut-document-understanding-transformer","slug":"donut-document-understanding-transformer","title":"OCR-free Document Understanding Transformer","date":"2021-11-30","arxiv_id":"2111.15664","repositories_listed":5,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/donut-document-understanding-transformer#ran","syntology_url":"https://syntology.ai/paper/2111.15664","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.15664"}},"official":{"repos":["clovaai/donut"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/classification-regression-for-chart","slug":"classification-regression-for-chart","title":"Classification-Regression for Chart Comprehension","date":"2021-11-29","arxiv_id":"2111.14792","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/classification-regression-for-chart#ran","syntology_url":"https://syntology.ai/paper/2111.14792","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.14792"}},"official":{"repos":["levymsn/cqa-crct"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crossing-the-format-boundary-of-text-and","slug":"crossing-the-format-boundary-of-text-and","title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","date":"2021-11-23","arxiv_id":"2111.12085","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crossing-the-format-boundary-of-text-and#ran","syntology_url":"https://syntology.ai/paper/2111.12085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12085"}},"official":{"repos":["microsoft/UniTAB"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-grained-vision-language-pre-training","slug":"multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","arxiv_id":"2111.08276","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-grained-vision-language-pre-training#ran","syntology_url":"https://syntology.ai/paper/2111.08276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08276"}},"official":{"repos":["zengyan-97/x-vlm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-of-training-end-to-end","slug":"an-empirical-study-of-training-end-to-end","title":"An Empirical Study of Training End-to-End Vision-and-Language Transformers","date":"2021-11-03","arxiv_id":"2111.02387","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-of-training-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2111.02387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.02387"}},"official":{"repos":["zdou0830/meter"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/introspective-distillation-for-robust","slug":"introspective-distillation-for-robust","title":"Introspective Distillation for Robust Question Answering","date":"2021-11-01","arxiv_id":"2111.01026","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":4,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/introspective-distillation-for-robust#ran","syntology_url":"https://syntology.ai/paper/2111.01026","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.01026"}},"official":{"repos":["yuleiniu/introd"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/perceptual-score-what-data-modalities-does","slug":"perceptual-score-what-data-modalities-does","title":"Perceptual Score: What Data Modalities Does Your Model Perceive?","date":"2021-10-27","arxiv_id":"2110.14375","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/perceptual-score-what-data-modalities-does#ran","syntology_url":"https://syntology.ai/paper/2110.14375","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.14375"}},"official":{"repos":["itaigat/perceptual-score"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alignment-attention-by-matching-key-and-query","slug":"alignment-attention-by-matching-key-and-query","title":"Alignment Attention by Matching Key and Query Distributions","date":"2021-10-25","arxiv_id":"2110.12567","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/alignment-attention-by-matching-key-and-query#ran","syntology_url":"https://syntology.ai/paper/2110.12567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.12567"}},"official":{"repos":["szhang42/alignment_attention"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/iconqa-a-new-benchmark-for-abstract-diagram","slug":"iconqa-a-new-benchmark-for-abstract-diagram","title":"IconQA: A New Benchmark for Abstract Diagram Understanding and Visual Language Reasoning","date":"2021-10-25","arxiv_id":"2110.13214","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/iconqa-a-new-benchmark-for-abstract-diagram#ran","syntology_url":"https://syntology.ai/paper/2110.13214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13214"}},"official":{"repos":["lupantech/iconqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/label-descriptive-patterns-and-their","slug":"label-descriptive-patterns-and-their","title":"Label-Descriptive Patterns and Their Application to Characterizing Classification Errors","date":"2021-10-18","arxiv_id":"2110.09599","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/label-descriptive-patterns-and-their#ran","syntology_url":"https://syntology.ai/paper/2110.09599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.09599"}},"official":{"repos":["uds-lsv/premise"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/coarse-to-fine-reasoning-for-visual-question","slug":"coarse-to-fine-reasoning-for-visual-question","title":"Coarse-to-Fine Reasoning for Visual Question Answering","date":"2021-10-06","arxiv_id":"2110.02526","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":2,"n_violates":1,"n_no_contract":3,"n_pointer_only":2,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/coarse-to-fine-reasoning-for-visual-question#ran","syntology_url":"https://syntology.ai/paper/2110.02526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.02526"}},"official":{"repos":["aioz-ai/cfr_vqa","aioz-ai/crf_vqa"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-empirical-study-of-gpt-3-for-few-shot","slug":"an-empirical-study-of-gpt-3-for-few-shot","title":"An Empirical Study of GPT-3 for Few-Shot Knowledge-Based VQA","date":"2021-09-10","arxiv_id":"2109.05014","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-of-gpt-3-for-few-shot#ran","syntology_url":"https://syntology.ai/paper/2109.05014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.05014"}},"official":{"repos":["microsoft/PICa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/weakly-supervised-visual-retriever-reader-for","slug":"weakly-supervised-visual-retriever-reader-for","title":"Weakly-Supervised Visual-Retriever-Reader for Knowledge-based Question Answering","date":"2021-09-09","arxiv_id":"2109.04014","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/weakly-supervised-visual-retriever-reader-for#ran","syntology_url":"https://syntology.ai/paper/2109.04014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.04014"}},"official":{"repos":["luomancs/retriever_reader_for_okvqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/webqa-multihop-and-multimodal-qa","slug":"webqa-multihop-and-multimodal-qa","title":"WebQA: Multihop and Multimodal QA","date":"2021-09-01","arxiv_id":"2109.00590","repositories_listed":3,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/webqa-multihop-and-multimodal-qa#ran","syntology_url":"https://syntology.ai/paper/2109.00590","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.00590"}},"official":null}},{"url":"/paper/simvlm-simple-visual-language-model","slug":"simvlm-simple-visual-language-model","title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision","date":"2021-08-24","arxiv_id":"2108.10904","repositories_listed":2,"syntology":{"n":37,"n_ran":22,"n_constructed":8,"n_ran_checked":16,"n_instrument":6,"n_unverified":15,"n_honours":1,"n_violates":3,"n_no_contract":12,"n_pointer_only":32,"phrase":"22 ran (of which 8 constructed an object rather than computing a result; 16 with no instrument failure: 1 honoured, 3 violated, 12 with no contract checked; 6 where Syntology's instrument failed) · 15 unverified","sample_list":"/paper/simvlm-simple-visual-language-model#ran","syntology_url":"https://syntology.ai/paper/2108.10904","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.10904"}},"official":null}},{"url":"/paper/sparse-continuous-distributions-and-fenchel","slug":"sparse-continuous-distributions-and-fenchel","title":"Sparse Continuous Distributions and Fenchel-Young Losses","date":"2021-08-04","arxiv_id":"2108.01988","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sparse-continuous-distributions-and-fenchel#ran","syntology_url":"https://syntology.ai/paper/2108.01988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.01988"}},"official":{"repos":["deep-spin/sparse_continuous_distributions"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/greedy-gradient-ensemble-for-robust-visual","slug":"greedy-gradient-ensemble-for-robust-visual","title":"Greedy Gradient Ensemble for Robust Visual Question Answering","date":"2021-07-27","arxiv_id":"2107.12651","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/greedy-gradient-ensemble-for-robust-visual#ran","syntology_url":"https://syntology.ai/paper/2107.12651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.12651"}},"official":{"repos":["GeraldHan/GGE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/separating-skills-and-concepts-for-novel-1","slug":"separating-skills-and-concepts-for-novel-1","title":"Separating Skills and Concepts for Novel Visual Question Answering","date":"2021-07-19","arxiv_id":"2107.09106","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/separating-skills-and-concepts-for-novel-1#ran","syntology_url":"https://syntology.ai/paper/2107.09106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.09106"}},"official":{"repos":["SpencerWhitehead/novelvqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/align-before-fuse-vision-and-language","slug":"align-before-fuse-vision-and-language","title":"Align before Fuse: Vision and Language Representation Learning with Momentum Distillation","date":"2021-07-16","arxiv_id":"2107.07651","repositories_listed":6,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/align-before-fuse-vision-and-language#ran","syntology_url":"https://syntology.ai/paper/2107.07651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.07651"}},"official":{"repos":["salesforce/lavis"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/how-much-can-clip-benefit-vision-and-language","slug":"how-much-can-clip-benefit-vision-and-language","title":"How Much Can CLIP Benefit Vision-and-Language Tasks?","date":"2021-07-13","arxiv_id":"2107.06383","repositories_listed":4,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":2,"n_instrument":5,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/how-much-can-clip-benefit-vision-and-language#ran","syntology_url":"https://syntology.ai/paper/2107.06383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.06383"}},"official":{"repos":["clip-vil/CLIP-ViL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/mind-your-outliers-investigating-the-negative","slug":"mind-your-outliers-investigating-the-negative","title":"Mind Your Outliers! Investigating the Negative Impact of Outliers on Active Learning for Visual Question Answering","date":"2021-07-06","arxiv_id":"2107.02331","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-your-outliers-investigating-the-negative#ran","syntology_url":"https://syntology.ai/paper/2107.02331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.02331"}},"official":{"repos":["siddk/vqa-outliers"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/naaqa-a-neural-architecture-for-acoustic","slug":"naaqa-a-neural-architecture-for-acoustic","title":"NAAQA: A Neural Architecture for Acoustic Question Answering","date":"2021-06-11","arxiv_id":"2106.06147","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/naaqa-a-neural-architecture-for-acoustic#ran","syntology_url":"https://syntology.ai/paper/2106.06147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06147"}},"official":null}},{"url":"/paper/multi-modal-understanding-and-generation-for","slug":"multi-modal-understanding-and-generation-for","title":"Multi-modal Understanding and Generation for Medical Images and Text via Vision-Language Pre-Training","date":"2021-05-24","arxiv_id":"2105.11333","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-modal-understanding-and-generation-for#ran","syntology_url":"https://syntology.ai/paper/2105.11333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.11333"}},"official":{"repos":["SuperSupermoon/MedViLL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multiple-meta-model-quantifying-for-medical","slug":"multiple-meta-model-quantifying-for-medical","title":"Multiple Meta-model Quantifying for Medical Visual Question Answering","date":"2021-05-19","arxiv_id":"2105.08913","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":3,"n_no_contract":4,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 3 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/multiple-meta-model-quantifying-for-medical#ran","syntology_url":"https://syntology.ai/paper/2105.08913","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.08913"}},"official":{"repos":["aioz-ai/MICCAI21_MMQ"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/next-qa-next-phase-of-question-answering-to","slug":"next-qa-next-phase-of-question-answering-to","title":"NExT-QA:Next Phase of Question-Answering to Explaining Temporal Actions","date":"2021-05-18","arxiv_id":"2105.08276","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/next-qa-next-phase-of-question-answering-to#ran","syntology_url":"https://syntology.ai/paper/2105.08276","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.08276"}},"official":{"repos":["doc-doc/NExT-QA"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inter-gps-interpretable-geometry-problem","slug":"inter-gps-interpretable-geometry-problem","title":"Inter-GPS: Interpretable Geometry Problem Solving with Formal Language and Symbolic Reasoning","date":"2021-05-10","arxiv_id":"2105.04165","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/inter-gps-interpretable-geometry-problem#ran","syntology_url":"https://syntology.ai/paper/2105.04165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.04165"}},"official":{"repos":["lupantech/InterGPS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/passage-retrieval-for-outside-knowledge","slug":"passage-retrieval-for-outside-knowledge","title":"Passage Retrieval for Outside-Knowledge Visual Question Answering","date":"2021-05-09","arxiv_id":"2105.03938","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/passage-retrieval-for-outside-knowledge#ran","syntology_url":"https://syntology.ai/paper/2105.03938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.03938"}},"official":{"repos":["prdwb/okvqa-release"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mdetr-modulated-detection-for-end-to-end","slug":"mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","arxiv_id":"2104.12763","repositories_listed":5,"syntology":{"n":11,"n_ran":7,"n_constructed":4,"n_ran_checked":6,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/mdetr-modulated-detection-for-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2104.12763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.12763"}},"official":{"repos":["ashkamath/mdetr"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/beyond-question-based-biases-assessing","slug":"beyond-question-based-biases-assessing","title":"Beyond Question-Based Biases: Assessing Multimodal Shortcut Learning in Visual Question Answering","date":"2021-04-07","arxiv_id":"2104.03149","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-question-based-biases-assessing#ran","syntology_url":"https://syntology.ai/paper/2104.03149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.03149"}},"official":{"repos":["cdancette/detect-shortcuts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/an-investigation-of-critical-issues-in-bias","slug":"an-investigation-of-critical-issues-in-bias","title":"Are Bias Mitigation Techniques for Deep Learning Effective?","date":"2021-04-01","arxiv_id":"2104.00170","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/an-investigation-of-critical-issues-in-bias#ran","syntology_url":"https://syntology.ai/paper/2104.00170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00170"}},"official":{"repos":["erobic/bias-mitigators"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-general-purpose-vision-systems","slug":"towards-general-purpose-vision-systems","title":"Towards General Purpose Vision Systems","date":"2021-04-01","arxiv_id":"2104.00743","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-general-purpose-vision-systems#ran","syntology_url":"https://syntology.ai/paper/2104.00743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00743"}},"official":{"repos":["allenai/gpv-1"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/trafficqa-a-question-answering-benchmark-and","slug":"trafficqa-a-question-answering-benchmark-and","title":"SUTD-TrafficQA: A Question Answering Benchmark and an Efficient Network for Video Reasoning over Traffic Events","date":"2021-03-29","arxiv_id":"2103.15538","repositories_listed":3,"syntology":{"n":9,"n_ran":6,"n_constructed":1,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/trafficqa-a-question-answering-benchmark-and#ran","syntology_url":"https://syntology.ai/paper/2103.15538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.15538"}},"official":{"repos":["SUTDCV/SUTD-TrafficQA"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/generic-attention-model-explainability-for","slug":"generic-attention-model-explainability-for","title":"Generic Attention-model Explainability for Interpreting Bi-Modal and Encoder-Decoder Transformers","date":"2021-03-29","arxiv_id":"2103.15679","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generic-attention-model-explainability-for#ran","syntology_url":"https://syntology.ai/paper/2103.15679","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.15679"}},"official":{"repos":["hila-chefer/Transformer-MM-Explainability"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-generation-of-contrast-sets-from","slug":"automatic-generation-of-contrast-sets-from","title":"Automatic Generation of Contrast Sets from Scene Graphs: Probing the Compositional Consistency of GQA","date":"2021-03-17","arxiv_id":"2103.09591","repositories_listed":2,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/automatic-generation-of-contrast-sets-from#ran","syntology_url":"https://syntology.ai/paper/2103.09591","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.09591"}},"official":{"repos":["yonatanbitton/AutoGenOfContrastSetsFromSceneGraphs"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/vilt-vision-and-language-transformer-without","slug":"vilt-vision-and-language-transformer-without","title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision","date":"2021-02-05","arxiv_id":"2102.03334","repositories_listed":6,"syntology":{"n":4,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/vilt-vision-and-language-transformer-without#ran","syntology_url":"https://syntology.ai/paper/2102.03334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.03334"}},"official":{"repos":["dandelin/vilt"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed"]}}},{"url":"/paper/unifying-vision-and-language-tasks-via-text","slug":"unifying-vision-and-language-tasks-via-text","title":"Unifying Vision-and-Language Tasks via Text Generation","date":"2021-02-04","arxiv_id":"2102.02779","repositories_listed":2,"syntology":{"n":12,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/unifying-vision-and-language-tasks-via-text#ran","syntology_url":"https://syntology.ai/paper/2102.02779","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.02779"}},"official":{"repos":["j-min/VL-T5"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/overcoming-language-priors-with-self","slug":"overcoming-language-priors-with-self","title":"Overcoming Language Priors with Self-supervised Learning for Visual Question Answering","date":"2020-12-17","arxiv_id":"2012.11528","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":3,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/overcoming-language-priors-with-self#ran","syntology_url":"https://syntology.ai/paper/2012.11528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.11528"}},"official":{"repos":["CrossmodalGroup/SSL-VQA"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-model-and-ignore-dataset-bias","slug":"learning-to-model-and-ignore-dataset-bias","title":"Learning to Model and Ignore Dataset Bias with Mixed Capacity Ensembles","date":"2020-11-07","arxiv_id":"2011.03856","repositories_listed":1,"syntology":{"n":18,"n_ran":10,"n_constructed":5,"n_ran_checked":8,"n_instrument":2,"n_unverified":8,"n_honours":2,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/learning-to-model-and-ignore-dataset-bias#ran","syntology_url":"https://syntology.ai/paper/2011.03856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.03856"}},"official":{"repos":["chrisc36/autobias"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":8,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/loss-rescaling-vqa-revisiting-language-prior","slug":"loss-rescaling-vqa-revisiting-language-prior","title":"Loss re-scaling VQA: Revisiting the LanguagePrior Problem from a Class-imbalance View","date":"2020-10-30","arxiv_id":"2010.16010","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/loss-rescaling-vqa-revisiting-language-prior#ran","syntology_url":"https://syntology.ai/paper/2010.16010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.16010"}},"official":{"repos":["guoyang9/class-imbalance-VQA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sort-ing-vqa-models-contrastive-gradient","slug":"sort-ing-vqa-models-contrastive-gradient","title":"SOrT-ing VQA Models : Contrastive Gradient Learning for Improved Consistency","date":"2020-10-20","arxiv_id":"2010.10038","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sort-ing-vqa-models-contrastive-gradient#ran","syntology_url":"https://syntology.ai/paper/2010.10038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10038"}},"official":{"repos":["sameerdharur/sorting-vqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/x-lxmert-paint-caption-and-answer-questions","slug":"x-lxmert-paint-caption-and-answer-questions","title":"X-LXMERT: Paint, Caption and Answer Questions with Multi-Modal Transformers","date":"2020-09-23","arxiv_id":"2009.11278","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/x-lxmert-paint-caption-and-answer-questions#ran","syntology_url":"https://syntology.ai/paper/2009.11278","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.11278"}},"official":{"repos":["allenai/x-lxmert"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-based-video-question-answering-with","slug":"knowledge-based-video-question-answering-with","title":"Knowledge-Based Video Question Answering with Unsupervised Scene Descriptions","date":"2020-07-17","arxiv_id":"2007.08751","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowledge-based-video-question-answering-with#ran","syntology_url":"https://syntology.ai/paper/2007.08751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.08751"}},"official":{"repos":["noagarcia/ROLL-VideoQA"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/docvqa-a-dataset-for-vqa-on-document-images","slug":"docvqa-a-dataset-for-vqa-on-document-images","title":"DocVQA: A Dataset for VQA on Document Images","date":"2020-07-01","arxiv_id":"2007.00398","repositories_listed":3,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/docvqa-a-dataset-for-vqa-on-document-images#ran","syntology_url":"https://syntology.ai/paper/2007.00398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.00398"}},"official":null}},{"url":"/paper/graph-optimal-transport-for-cross-domain","slug":"graph-optimal-transport-for-cross-domain","title":"Graph Optimal Transport for Cross-Domain Alignment","date":"2020-06-26","arxiv_id":"2006.14744","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":2,"n_no_contract":6,"n_pointer_only":3,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 2 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/graph-optimal-transport-for-cross-domain#ran","syntology_url":"https://syntology.ai/paper/2006.14744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.14744"}},"official":{"repos":["LiqunChen0606/Graph-Optimal-Transport"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neuro-symbolic-visual-reasoning-disentangling","slug":"neuro-symbolic-visual-reasoning-disentangling","title":"Neuro-Symbolic Visual Reasoning: Disentangling \"Visual\" from \"Reasoning\"","date":"2020-06-20","arxiv_id":"2006.11524","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/neuro-symbolic-visual-reasoning-disentangling#ran","syntology_url":"https://syntology.ai/paper/2006.11524","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.11524"}},"official":null}},{"url":"/paper/sparse-and-continuous-attention-mechanisms","slug":"sparse-and-continuous-attention-mechanisms","title":"Sparse and Continuous Attention Mechanisms","date":"2020-06-12","arxiv_id":"2006.07214","repositories_listed":2,"syntology":{"n":15,"n_ran":12,"n_constructed":9,"n_ran_checked":9,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"12 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sparse-and-continuous-attention-mechanisms#ran","syntology_url":"https://syntology.ai/paper/2006.07214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.07214"}},"official":{"repos":["deep-spin/mcan-vqa-continuous-attention"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["listed"]}}},{"url":"/paper/large-scale-adversarial-training-for-vision","slug":"large-scale-adversarial-training-for-vision","title":"Large-Scale Adversarial Training for Vision-and-Language Representation Learning","date":"2020-06-11","arxiv_id":"2006.06195","repositories_listed":2,"syntology":{"n":20,"n_ran":12,"n_constructed":5,"n_ran_checked":9,"n_instrument":3,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":8,"phrase":"12 ran (of which 5 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/large-scale-adversarial-training-for-vision#ran","syntology_url":"https://syntology.ai/paper/2006.06195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06195"}},"official":{"repos":["zhegan27/LXMERT-AdvTrain","zhegan27/VILLA"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":5,"n_ran_no_instrument_failure":9,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/closed-loop-neural-symbolic-learning-via","slug":"closed-loop-neural-symbolic-learning-via","title":"Closed Loop Neural-Symbolic Learning via Integrating Neural Perception, Grammar Parsing, and Symbolic Reasoning","date":"2020-06-11","arxiv_id":"2006.06649","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/closed-loop-neural-symbolic-learning-via#ran","syntology_url":"https://syntology.ai/paper/2006.06649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.06649"}},"official":{"repos":["liqing-ustc/NGS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cobra-contrastive-bi-modal-representation","slug":"cobra-contrastive-bi-modal-representation","title":"COBRA: Contrastive Bi-Modal Representation Algorithm","date":"2020-05-07","arxiv_id":"2005.03687","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cobra-contrastive-bi-modal-representation#ran","syntology_url":"https://syntology.ai/paper/2005.03687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.03687"}},"official":{"repos":["ovshake/cobra"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-language-binding-in-relational-visual","slug":"dynamic-language-binding-in-relational-visual","title":"Dynamic Language Binding in Relational Visual Reasoning","date":"2020-04-30","arxiv_id":"2004.14603","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":5,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/dynamic-language-binding-in-relational-visual#ran","syntology_url":"https://syntology.ai/paper/2004.14603","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14603"}},"official":null}},{"url":"/paper/pragmatic-issue-sensitive-image-captioning","slug":"pragmatic-issue-sensitive-image-captioning","title":"Pragmatic Issue-Sensitive Image Captioning","date":"2020-04-29","arxiv_id":"2004.14451","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pragmatic-issue-sensitive-image-captioning#ran","syntology_url":"https://syntology.ai/paper/2004.14451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.14451"}},"official":{"repos":["windweller/Pragmatic-ISIC"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/oscar-object-semantics-aligned-pre-training","slug":"oscar-object-semantics-aligned-pre-training","title":"Oscar: Object-Semantics Aligned Pre-training for Vision-Language Tasks","date":"2020-04-13","arxiv_id":"2004.06165","repositories_listed":4,"syntology":{"n":23,"n_ran":13,"n_constructed":6,"n_ran_checked":10,"n_instrument":3,"n_unverified":10,"n_honours":1,"n_violates":0,"n_no_contract":9,"n_pointer_only":1,"phrase":"13 ran (of which 6 constructed an object rather than computing a result; 10 with no instrument failure: 1 honoured, 0 violated, 9 with no contract checked; 3 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/oscar-object-semantics-aligned-pre-training#ran","syntology_url":"https://syntology.ai/paper/2004.06165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.06165"}},"official":{"repos":["microsoft/Oscar"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-negative-case-analysis-of-visual-grounding","slug":"a-negative-case-analysis-of-visual-grounding","title":"Visual Grounding Methods for VQA are Working for the Wrong Reasons!","date":"2020-04-12","arxiv_id":"2004.05704","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-negative-case-analysis-of-visual-grounding#ran","syntology_url":"https://syntology.ai/paper/2004.05704","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.05704"}},"official":{"repos":["erobic/negative_analysis_of_grounding"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-ground-truth-evaluation-of-visual","slug":"towards-ground-truth-evaluation-of-visual","title":"Ground Truth Evaluation of Neural Network Explanations with CLEVR-XAI","date":"2020-03-16","arxiv_id":"2003.07258","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-ground-truth-evaluation-of-visual#ran","syntology_url":"https://syntology.ai/paper/2003.07258","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.07258"}},"official":{"repos":["ahmedmagdiosman/clevr-xai","ahmedmagdiosman/simply-clevr-dataset"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-samples-synthesizing-for","slug":"counterfactual-samples-synthesizing-for","title":"Counterfactual Samples Synthesizing for Robust Visual Question Answering","date":"2020-03-14","arxiv_id":"2003.06576","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-samples-synthesizing-for#ran","syntology_url":"https://syntology.ai/paper/2003.06576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06576"}},"official":{"repos":["yanxinzju/CSS-VQA"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/pathvqa-30000-questions-for-medical-visual","slug":"pathvqa-30000-questions-for-medical-visual","title":"PathVQA: 30000+ Questions for Medical Visual Question Answering","date":"2020-03-07","arxiv_id":"2003.10286","repositories_listed":5,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pathvqa-30000-questions-for-medical-visual#ran","syntology_url":"https://syntology.ai/paper/2003.10286","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.10286"}},"official":null}}],"record_sha256":"d675f0f28176ad7f3ffbd88531622be330070e611023ed91d6eab2d95bb2b8cf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}