{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering-1/papers/9","list_of":"/task/visual-question-answering-1","task":"Visual Question Answering","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":22,"rows_per_page":100,"rows":[801,900],"of":2177,"counts":{"archive_papers_tagged":2177,"with_a_code_link":1042,"where_syntology_ran_a_sample":378,"not_listed_spam_title":0,"listed":2177,"listed_where_code_ran":378,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":308,"every_run_a_failure_of_syntologys_instrument":70,"listed_with_a_run_with_no_instrument_failure":308,"listed_every_run_a_failure_of_syntologys_instrument":70,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering-1","prev":"/task/visual-question-answering-1/papers/8","next":"/task/visual-question-answering-1/papers/10","papers":[{"url":"/paper/a-okvqa-a-benchmark-for-visual-question","slug":"a-okvqa-a-benchmark-for-visual-question","title":"A-OKVQA: A Benchmark for Visual Question Answering using World Knowledge","date":"2022-06-03","arxiv_id":"2206.01718","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-okvqa-a-benchmark-for-visual-question#ran","syntology_url":"https://syntology.ai/paper/2206.01718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01718"}},"official":{"repos":["allenai/aokvqa"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/revive-regional-visual-representation-matters","slug":"revive-regional-visual-representation-matters","title":"REVIVE: Regional Visual Representation Matters in Knowledge-Based Visual Question Answering","date":"2022-06-02","arxiv_id":"2206.01201","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/revive-regional-visual-representation-matters#ran","syntology_url":"https://syntology.ai/paper/2206.01201","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.01201"}},"official":{"repos":["yzleroy/revive"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/expressive-scene-graph-generation-using","slug":"expressive-scene-graph-generation-using","title":"Expressive Scene Graph Generation Using Commonsense Knowledge Infusion for Visual Understanding and Reasoning","date":"2022-05-31","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-neuro-symbolic-asp-pipeline-for-visual","slug":"a-neuro-symbolic-asp-pipeline-for-visual","title":"A Neuro-Symbolic ASP Pipeline for Visual Question Answering","date":"2022-05-16","arxiv_id":"2205.07548","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-answer-visual-questions-from-web","slug":"learning-to-answer-visual-questions-from-web","title":"Learning to Answer Visual Questions from Web Videos","date":"2022-05-10","arxiv_id":"2205.05019","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/learning-to-answer-visual-questions-from-web#ran","syntology_url":"https://syntology.ai/paper/2205.05019","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.05019"}},"official":{"repos":["antoyang/just-ask"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/qlevr-a-diagnostic-dataset-for","slug":"qlevr-a-diagnostic-dataset-for","title":"QLEVR: A Diagnostic Dataset for Quantificational Language and Elementary Visual Reasoning","date":"2022-05-06","arxiv_id":"2205.03075","repositories_listed":1,"syntology":null},{"url":"/paper/declaration-based-prompt-tuning-for-visual","slug":"declaration-based-prompt-tuning-for-visual","title":"Declaration-based Prompt Tuning for Visual Question Answering","date":"2022-05-05","arxiv_id":"2205.02456","repositories_listed":1,"syntology":null},{"url":"/paper/what-is-right-for-me-is-not-yet-right-for-you","slug":"what-is-right-for-me-is-not-yet-right-for-you","title":"What is Right for Me is Not Yet Right for You: A Dataset for Grounding Relative Directions via Multi-Task Learning","date":"2022-05-05","arxiv_id":"2205.02671","repositories_listed":1,"syntology":null},{"url":"/paper/textrm-dureader-textrm-vis-a-chinese-dataset","slug":"textrm-dureader-textrm-vis-a-chinese-dataset","title":"\\textrm{DuReader}_{\\textrm{vis}}: A Chinese Dataset for Open-domain Document Visual Question Answering","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/vilmedic-a-framework-for-research-at-the","slug":"vilmedic-a-framework-for-research-at-the","title":"ViLMedic: a framework for research at the intersection of vision and language in medical AI","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/grit-general-robust-image-task-benchmark","slug":"grit-general-robust-image-task-benchmark","title":"GRIT: General Robust Image Task Benchmark","date":"2022-04-28","arxiv_id":"2204.13653","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grit-general-robust-image-task-benchmark#ran","syntology_url":"https://syntology.ai/paper/2204.13653","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13653"}},"official":{"repos":["allenai/grit_official"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reliable-visual-question-answering-abstain","slug":"reliable-visual-question-answering-abstain","title":"Reliable Visual Question Answering: Abstain Rather Than Answer Incorrectly","date":"2022-04-28","arxiv_id":"2204.13631","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reliable-visual-question-answering-abstain#ran","syntology_url":"https://syntology.ai/paper/2204.13631","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.13631"}},"official":{"repos":["facebookresearch/reliable_vqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hypergraph-transformer-weakly-supervised","slug":"hypergraph-transformer-weakly-supervised","title":"Hypergraph Transformer: Weakly-supervised Multi-hop Reasoning for Knowledge-based Visual Question Answering","date":"2022-04-22","arxiv_id":"2204.10448","repositories_listed":1,"syntology":null},{"url":"/paper/attention-in-reasoning-dataset-analysis-and","slug":"attention-in-reasoning-dataset-analysis-and","title":"Attention in Reasoning: Dataset, Analysis, and Modeling","date":"2022-04-20","arxiv_id":"2204.09774","repositories_listed":1,"syntology":null},{"url":"/paper/clevr-x-a-visual-reasoning-dataset-for","slug":"clevr-x-a-visual-reasoning-dataset-for","title":"CLEVR-X: A Visual Reasoning Dataset for Natural Language Explanations","date":"2022-04-05","arxiv_id":"2204.02380","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/clevr-x-a-visual-reasoning-dataset-for#ran","syntology_url":"https://syntology.ai/paper/2204.02380","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02380"}},"official":{"repos":["explainableml/clevr-x"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/swapmix-diagnosing-and-regularizing-the-over","slug":"swapmix-diagnosing-and-regularizing-the-over","title":"SwapMix: Diagnosing and Regularizing the Over-Reliance on Visual Context in Visual Question Answering","date":"2022-04-05","arxiv_id":"2204.02285","repositories_listed":1,"syntology":null},{"url":"/paper/vl-interpret-an-interactive-visualization","slug":"vl-interpret-an-interactive-visualization","title":"VL-InterpreT: An Interactive Visualization Tool for Interpreting Vision-Language Transformers","date":"2022-03-30","arxiv_id":"2203.17247","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vl-interpret-an-interactive-visualization#ran","syntology_url":"https://syntology.ai/paper/2203.17247","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.17247"}},"official":{"repos":["intellabs/vl-interpret"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/single-stream-multi-level-alignment-for","slug":"single-stream-multi-level-alignment-for","title":"Single-Stream Multi-Level Alignment for Vision-Language Pretraining","date":"2022-03-27","arxiv_id":"2203.14395","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-answer-questions-in-dynamic-audio","slug":"learning-to-answer-questions-in-dynamic-audio","title":"Learning to Answer Questions in Dynamic Audio-Visual Scenarios","date":"2022-03-26","arxiv_id":"2203.14072","repositories_listed":1,"syntology":{"n":4,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 4 unverified","sample_list":"/paper/learning-to-answer-questions-in-dynamic-audio#ran","syntology_url":"https://syntology.ai/paper/2203.14072","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.14072"}},"official":{"repos":["GeWu-Lab/MUSIC-AVQA"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"url":"/paper/a-stitch-in-time-saves-nine-a-train-time","slug":"a-stitch-in-time-saves-nine-a-train-time","title":"A Stitch in Time Saves Nine: A Train-Time Regularizing Loss for Improved Neural Network Calibration","date":"2022-03-25","arxiv_id":"2203.13834","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/a-stitch-in-time-saves-nine-a-train-time#ran","syntology_url":"https://syntology.ai/paper/2203.13834","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.13834"}},"official":{"repos":["mdca-loss/mdca-calibration"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-efficient-and-elastic-visual-question","slug":"towards-efficient-and-elastic-visual-question","title":"Bilaterally Slimmable Transformer for Elastic and Efficient Visual Question Answering","date":"2022-03-24","arxiv_id":"2203.12814","repositories_listed":1,"syntology":null},{"url":"/paper/mukea-multimodal-knowledge-extraction-and","slug":"mukea-multimodal-knowledge-extraction-and","title":"MuKEA: Multimodal Knowledge Extraction and Accumulation for Knowledge-based Visual Question Answering","date":"2022-03-17","arxiv_id":"2203.09138","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mukea-multimodal-knowledge-extraction-and#ran","syntology_url":"https://syntology.ai/paper/2203.09138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.09138"}},"official":{"repos":["andersonstra/mukea"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/barlow-constrained-optimization-for-visual","slug":"barlow-constrained-optimization-for-visual","title":"Barlow constrained optimization for Visual Question Answering","date":"2022-03-07","arxiv_id":"2203.03727","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-key-value-memory-enhanced-multi-step","slug":"dynamic-key-value-memory-enhanced-multi-step","title":"Dynamic Key-value Memory Enhanced Multi-step Graph Reasoning for Knowledge-based Visual Question Answering","date":"2022-03-06","arxiv_id":"2203.02985","repositories_listed":1,"syntology":null},{"url":"/paper/joint-answering-and-explanation-for-visual","slug":"joint-answering-and-explanation-for-visual","title":"Joint Answering and Explanation for Visual Commonsense Reasoning","date":"2022-02-25","arxiv_id":"2202.12626","repositories_listed":1,"syntology":null},{"url":"/paper/on-modality-bias-recognition-and-reduction","slug":"on-modality-bias-recognition-and-reduction","title":"On Modality Bias Recognition and Reduction","date":"2022-02-25","arxiv_id":"2202.12690","repositories_listed":1,"syntology":null},{"url":"/paper/og-sgg-ontology-guided-scene-graph-generation","slug":"og-sgg-ontology-guided-scene-graph-generation","title":"OG-SGG: Ontology-Guided Scene Graph Generation. A Case Study in Transfer Learning for Telepresence Robotics","date":"2022-02-21","arxiv_id":"2202.10201","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-pre-training-with-triple","slug":"vision-language-pre-training-with-triple","title":"Vision-Language Pre-Training with Triple Contrastive Learning","date":"2022-02-21","arxiv_id":"2202.10401","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/vision-language-pre-training-with-triple#ran","syntology_url":"https://syntology.ai/paper/2202.10401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10401"}},"official":{"repos":["uta-smile/TCL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/delving-deeper-into-cross-lingual-visual","slug":"delving-deeper-into-cross-lingual-visual","title":"Delving Deeper into Cross-lingual Visual Question Answering","date":"2022-02-15","arxiv_id":"2202.07630","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-answers-for-visual-questions-asked","slug":"grounding-answers-for-visual-questions-asked","title":"Grounding Answers for Visual Questions Asked by Visually Impaired People","date":"2022-02-04","arxiv_id":"2202.01993","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounding-answers-for-visual-questions-asked#ran","syntology_url":"https://syntology.ai/paper/2202.01993","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.01993"}},"official":{"repos":["ccychongyanchen/vizwizvqagroundingcrowdsourcing"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/compositionality-as-lexical-symmetry","slug":"compositionality-as-lexical-symmetry","title":"Compositionality as Lexical Symmetry","date":"2022-01-30","arxiv_id":"2201.12926","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/compositionality-as-lexical-symmetry#ran","syntology_url":"https://syntology.ai/paper/2201.12926","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.12926"}},"official":{"repos":["ekinakyurek/lexsym"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/transformer-module-networks-for-systematic","slug":"transformer-module-networks-for-systematic","title":"Transformer Module Networks for Systematic Generalization in Visual Question Answering","date":"2022-01-27","arxiv_id":"2201.11316","repositories_listed":1,"syntology":null},{"url":"/paper/maintaining-reasoning-consistency-in","slug":"maintaining-reasoning-consistency-in","title":"Maintaining Reasoning Consistency in Compositional Visual Question Answering","date":"2022-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/query-and-attention-augmentation-for","slug":"query-and-attention-augmentation-for","title":"Query and Attention Augmentation for Knowledge-Based Explainable Reasoning","date":"2022-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-image-visual-question-answering","slug":"multi-image-visual-question-answering","title":"Multi-Image Visual Question Answering","date":"2021-12-27","arxiv_id":"2112.13706","repositories_listed":1,"syntology":null},{"url":"/paper/latr-layout-aware-transformer-for-scene-text","slug":"latr-layout-aware-transformer-for-scene-text","title":"LaTr: Layout-Aware Transformer for Scene-Text VQA","date":"2021-12-23","arxiv_id":"2112.12494","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/latr-layout-aware-transformer-for-scene-text#ran","syntology_url":"https://syntology.ai/paper/2112.12494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.12494"}},"official":null}},{"url":"/paper/clevr3d-compositional-language-and-elementary","slug":"clevr3d-compositional-language-and-elementary","title":"Comprehensive Visual Question Answering on Point Clouds through Compositional Scene Manipulation","date":"2021-12-22","arxiv_id":"2112.11691","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clevr3d-compositional-language-and-elementary#ran","syntology_url":"https://syntology.ai/paper/2112.11691","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.11691"}},"official":{"repos":["yanx27/clevr3d"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/general-greedy-de-bias-learning","slug":"general-greedy-de-bias-learning","title":"General Greedy De-bias Learning","date":"2021-12-20","arxiv_id":"2112.10572","repositories_listed":1,"syntology":null},{"url":"/paper/bilateral-cross-modality-graph-matching","slug":"bilateral-cross-modality-graph-matching","title":"Bilateral Cross-Modality Graph Matching Attention for Feature Fusion in Visual Question Answering","date":"2021-12-14","arxiv_id":"2112.07270","repositories_listed":1,"syntology":null},{"url":"/paper/dual-key-multimodal-backdoors-for-visual","slug":"dual-key-multimodal-backdoors-for-visual","title":"Dual-Key Multimodal Backdoors for Visual Question Answering","date":"2021-12-14","arxiv_id":"2112.07668","repositories_listed":1,"syntology":null},{"url":"/paper/change-detection-meets-visual-question","slug":"change-detection-meets-visual-question","title":"Change Detection Meets Visual Question Answering","date":"2021-12-12","arxiv_id":"2112.06343","repositories_listed":1,"syntology":null},{"url":"/paper/debiased-visual-question-answering-from","slug":"debiased-visual-question-answering-from","title":"Debiased Visual Question Answering from Feature and Sample Perspectives","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/crossing-the-format-boundary-of-text-and","slug":"crossing-the-format-boundary-of-text-and","title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","date":"2021-11-23","arxiv_id":"2111.12085","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":9,"n_instrument":5,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":5,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/crossing-the-format-boundary-of-text-and#ran","syntology_url":"https://syntology.ai/paper/2111.12085","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.12085"}},"official":{"repos":["microsoft/UniTAB"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/many-heads-but-one-brain-an-overview-of","slug":"many-heads-but-one-brain-an-overview-of","title":"Many Heads but One Brain: Fusion Brain -- a Competition and a Single Multimodal Multitask Architecture","date":"2021-11-22","arxiv_id":"2111.10974","repositories_listed":1,"syntology":null},{"url":"/paper/mirtt-learning-multimodal-interaction","slug":"mirtt-learning-multimodal-interaction","title":"MIRTT: Learning Multimodal Interaction Representations from Trilinear Transformers for Visual Question Answering","date":"2021-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/alignment-attention-by-matching-key-and-query","slug":"alignment-attention-by-matching-key-and-query","title":"Alignment Attention by Matching Key and Query Distributions","date":"2021-10-25","arxiv_id":"2110.12567","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/alignment-attention-by-matching-key-and-query#ran","syntology_url":"https://syntology.ai/paper/2110.12567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.12567"}},"official":{"repos":["szhang42/alignment_attention"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/iconqa-a-new-benchmark-for-abstract-diagram","slug":"iconqa-a-new-benchmark-for-abstract-diagram","title":"IconQA: A New Benchmark for Abstract Diagram Understanding and Visual Language Reasoning","date":"2021-10-25","arxiv_id":"2110.13214","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/iconqa-a-new-benchmark-for-abstract-diagram#ran","syntology_url":"https://syntology.ai/paper/2110.13214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.13214"}},"official":{"repos":["lupantech/iconqa"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dair-data-augmented-invariant-regularization-1","slug":"dair-data-augmented-invariant-regularization-1","title":"Robustness through Data Augmentation Loss Consistency","date":"2021-10-21","arxiv_id":"2110.11205","repositories_listed":1,"syntology":null},{"url":"/paper/towards-language-guided-visual-recognition","slug":"towards-language-guided-visual-recognition","title":"Towards Language-guided Visual Recognition via Dynamic Convolutions","date":"2021-10-17","arxiv_id":"2110.08797","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-accuracy-a-consolidated-tool-for","slug":"beyond-accuracy-a-consolidated-tool-for","title":"Beyond Accuracy: A Consolidated Tool for Visual Question Answering Benchmarking","date":"2021-10-11","arxiv_id":"2110.05159","repositories_listed":1,"syntology":null},{"url":"/paper/pano-avqa-grounded-audio-visual-question-1","slug":"pano-avqa-grounded-audio-visual-question-1","title":"Pano-AVQA: Grounded Audio-Visual Question Answering on 360$^\\circ$ Videos","date":"2021-10-11","arxiv_id":"2110.05122","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-samples-synthesizing-and","slug":"counterfactual-samples-synthesizing-and","title":"Counterfactual Samples Synthesizing and Training for Robust Visual Question Answering","date":"2021-10-03","arxiv_id":"2110.01013","repositories_listed":1,"syntology":null},{"url":"/paper/calibrating-concepts-and-operations-towards","slug":"calibrating-concepts-and-operations-towards","title":"Calibrating Concepts and Operations: Towards Symbolic Reasoning on Real Images","date":"2021-10-01","arxiv_id":"2110.00519","repositories_listed":1,"syntology":null},{"url":"/paper/the-spoon-is-in-the-sink-assisting-visually","slug":"the-spoon-is-in-the-sink-assisting-visually","title":"The Spoon Is in the Sink: Assisting Visually Impaired People in the Kitchen","date":"2021-10-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/does-vision-and-language-pretraining-improve","slug":"does-vision-and-language-pretraining-improve","title":"Does Vision-and-Language Pretraining Improve Lexical Grounding?","date":"2021-09-21","arxiv_id":"2109.10246","repositories_listed":1,"syntology":null},{"url":"/paper/image-captioning-for-effective-use-of","slug":"image-captioning-for-effective-use-of","title":"Image Captioning for Effective Use of Language Models in Knowledge-Based Visual Question Answering","date":"2021-09-15","arxiv_id":"2109.08029","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-the-unknown-knowns-turning","slug":"discovering-the-unknown-knowns-turning","title":"Discovering the Unknown Knowns: Turning Implicit Knowledge in the Dataset into Explicit Training Examples for Visual Question Answering","date":"2021-09-13","arxiv_id":"2109.06122","repositories_listed":1,"syntology":null},{"url":"/paper/xgqa-cross-lingual-visual-question-answering","slug":"xgqa-cross-lingual-visual-question-answering","title":"xGQA: Cross-Lingual Visual Question Answering","date":"2021-09-13","arxiv_id":"2109.06082","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-gpt-3-for-few-shot","slug":"an-empirical-study-of-gpt-3-for-few-shot","title":"An Empirical Study of GPT-3 for Few-Shot Knowledge-Based VQA","date":"2021-09-10","arxiv_id":"2109.05014","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-of-gpt-3-for-few-shot#ran","syntology_url":"https://syntology.ai/paper/2109.05014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.05014"}},"official":{"repos":["microsoft/PICa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improved-ramen-towards-domain-generalization","slug":"improved-ramen-towards-domain-generalization","title":"Improved RAMEN: Towards Domain Generalization for Visual Question Answering","date":"2021-09-06","arxiv_id":"2109.02370","repositories_listed":1,"syntology":null},{"url":"/paper/task-oriented-multi-user-semantic","slug":"task-oriented-multi-user-semantic","title":"Task-Oriented Multi-User Semantic Communications for VQA Task","date":"2021-08-16","arxiv_id":"2108.07357","repositories_listed":1,"syntology":null},{"url":"/paper/berthop-an-effective-vision-and-language","slug":"berthop-an-effective-vision-and-language","title":"BERTHop: An Effective Vision-and-Language Model for Chest X-ray Disease Diagnosis","date":"2021-08-10","arxiv_id":"2108.04938","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-continuous-distributions-and-fenchel","slug":"sparse-continuous-distributions-and-fenchel","title":"Sparse Continuous Distributions and Fenchel-Young Losses","date":"2021-08-04","arxiv_id":"2108.01988","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sparse-continuous-distributions-and-fenchel#ran","syntology_url":"https://syntology.ai/paper/2108.01988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.01988"}},"official":{"repos":["deep-spin/sparse_continuous_distributions"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/check-it-again-progressive-visual-question-1","slug":"check-it-again-progressive-visual-question-1","title":"Check It Again:Progressive Visual Question Answering via Visual Entailment","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/greedy-gradient-ensemble-for-robust-visual","slug":"greedy-gradient-ensemble-for-robust-visual","title":"Greedy Gradient Ensemble for Robust Visual Question Answering","date":"2021-07-27","arxiv_id":"2107.12651","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/greedy-gradient-ensemble-for-robust-visual#ran","syntology_url":"https://syntology.ai/paper/2107.12651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.12651"}},"official":{"repos":["GeraldHan/GGE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/x-ggm-graph-generative-modeling-for-out-of","slug":"x-ggm-graph-generative-modeling-for-out-of","title":"X-GGM: Graph Generative Modeling for Out-of-Distribution Generalization in Visual Question Answering","date":"2021-07-24","arxiv_id":"2107.11576","repositories_listed":1,"syntology":null},{"url":"/paper/graphhopper-multi-hop-scene-graph-reasoning","slug":"graphhopper-multi-hop-scene-graph-reasoning","title":"Graphhopper: Multi-Hop Scene Graph Reasoning for Visual Question Answering","date":"2021-07-13","arxiv_id":"2107.06325","repositories_listed":1,"syntology":null},{"url":"/paper/mind-your-outliers-investigating-the-negative","slug":"mind-your-outliers-investigating-the-negative","title":"Mind Your Outliers! Investigating the Negative Impact of Outliers on Active Learning for Visual Question Answering","date":"2021-07-06","arxiv_id":"2107.02331","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-your-outliers-investigating-the-negative#ran","syntology_url":"https://syntology.ai/paper/2107.02331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.02331"}},"official":{"repos":["siddk/vqa-outliers"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cognitive-visual-commonsense-reasoning-using","slug":"cognitive-visual-commonsense-reasoning-using","title":"Cognitive Visual Commonsense Reasoning Using Dynamic Working Memory","date":"2021-07-04","arxiv_id":"2107.01671","repositories_listed":1,"syntology":null},{"url":"/paper/perception-matters-detecting-perception","slug":"perception-matters-detecting-perception","title":"Perception Matters: Detecting Perception Failures of VQA Models Using Metamorphic Testing","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/predicting-human-scanpaths-in-visual-question","slug":"predicting-human-scanpaths-in-visual-question","title":"Predicting Human Scanpaths in Visual Question Answering","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rstnet-captioning-with-adaptive-attention-on","slug":"rstnet-captioning-with-adaptive-attention-on","title":"RSTNet: Captioning With Adaptive Attention on Visual and Non-Visual Words","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/probing-image-language-transformers-for-verb","slug":"probing-image-language-transformers-for-verb","title":"Probing Image-Language Transformers for Verb Understanding","date":"2021-06-16","arxiv_id":"2106.09141","repositories_listed":1,"syntology":null},{"url":"/paper/how-modular-should-neural-module-networks-be","slug":"how-modular-should-neural-module-networks-be","title":"How Modular Should Neural Module Networks Be for Systematic Generalization?","date":"2021-06-15","arxiv_id":"2106.08170","repositories_listed":1,"syntology":null},{"url":"/paper/naaqa-a-neural-architecture-for-acoustic","slug":"naaqa-a-neural-architecture-for-acoustic","title":"NAAQA: A Neural Architecture for Acoustic Question Answering","date":"2021-06-11","arxiv_id":"2106.06147","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/naaqa-a-neural-architecture-for-acoustic#ran","syntology_url":"https://syntology.ai/paper/2106.06147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06147"}},"official":null}},{"url":"/paper/check-it-again-progressive-visual-question","slug":"check-it-again-progressive-visual-question","title":"Check It Again: Progressive Visual Question Answering via Visual Entailment","date":"2021-06-08","arxiv_id":"2106.04605","repositories_listed":1,"syntology":null},{"url":"/paper/clevr-hyp-a-challenge-dataset-and-baselines-1","slug":"clevr-hyp-a-challenge-dataset-and-baselines-1","title":"CLEVR\\_HYP: A Challenge Dataset and Baselines for Visual Question Answering with Hypothetical Actions over Images","date":"2021-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ease-a-diagnostic-tool-for-vqa-based-on","slug":"ease-a-diagnostic-tool-for-vqa-based-on","title":"EaSe: A Diagnostic Tool for VQA based on Answer Diversity","date":"2021-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lpf-a-language-prior-feedback-objective","slug":"lpf-a-language-prior-feedback-objective","title":"LPF: A Language-Prior Feedback Objective Function for De-biased Visual Question Answering","date":"2021-05-29","arxiv_id":"2105.14300","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-understanding-and-generation-for","slug":"multi-modal-understanding-and-generation-for","title":"Multi-modal Understanding and Generation for Medical Images and Text via Vision-Language Pre-Training","date":"2021-05-24","arxiv_id":"2105.11333","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-modal-understanding-and-generation-for#ran","syntology_url":"https://syntology.ai/paper/2105.11333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.11333"}},"official":{"repos":["SuperSupermoon/MedViLL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/structurallm-structural-pre-training-for-form","slug":"structurallm-structural-pre-training-for-form","title":"StructuralLM: Structural Pre-training for Form Understanding","date":"2021-05-24","arxiv_id":"2105.11210","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 5 unverified","sample_list":"/paper/structurallm-structural-pre-training-for-form#ran","syntology_url":"https://syntology.ai/paper/2105.11210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.11210"}},"official":{"repos":["alibaba/AliceMind"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"url":"/paper/found-a-reason-for-me-weakly-supervised","slug":"found-a-reason-for-me-weakly-supervised","title":"Found a Reason for me? Weakly-supervised Grounded Visual Question Answering using Capsules","date":"2021-05-11","arxiv_id":"2105.04836","repositories_listed":1,"syntology":null},{"url":"/paper/passage-retrieval-for-outside-knowledge","slug":"passage-retrieval-for-outside-knowledge","title":"Passage Retrieval for Outside-Knowledge Visual Question Answering","date":"2021-05-09","arxiv_id":"2105.03938","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/passage-retrieval-for-outside-knowledge#ran","syntology_url":"https://syntology.ai/paper/2105.03938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.03938"}},"official":{"repos":["prdwb/okvqa-release"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adavqa-overcoming-language-priors-with","slug":"adavqa-overcoming-language-priors-with","title":"AdaVQA: Overcoming Language Priors with Adapted Margin Cosine Loss","date":"2021-05-05","arxiv_id":"2105.01993","repositories_listed":1,"syntology":null},{"url":"/paper/graghvqa-language-guided-graph-neural","slug":"graghvqa-language-guided-graph-neural","title":"GraghVQA: Language-Guided Graph Neural Networks for Graph-based Visual Question Answering","date":"2021-04-20","arxiv_id":"2104.10283","repositories_listed":1,"syntology":null},{"url":"/paper/clevr-hyp-a-challenge-dataset-and-baselines","slug":"clevr-hyp-a-challenge-dataset-and-baselines","title":"CLEVR_HYP: A Challenge Dataset and Baselines for Visual Question Answering with Hypothetical Actions over Images","date":"2021-04-13","arxiv_id":"2104.05981","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-question-based-biases-assessing","slug":"beyond-question-based-biases-assessing","title":"Beyond Question-Based Biases: Assessing Multimodal Shortcut Learning in Visual Question Answering","date":"2021-04-07","arxiv_id":"2104.03149","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-question-based-biases-assessing#ran","syntology_url":"https://syntology.ai/paper/2104.03149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.03149"}},"official":{"repos":["cdancette/detect-shortcuts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mmbert-multimodal-bert-pretraining-for","slug":"mmbert-multimodal-bert-pretraining-for","title":"MMBERT: Multimodal BERT Pretraining for Improved Medical VQA","date":"2021-04-03","arxiv_id":"2104.01394","repositories_listed":1,"syntology":null},{"url":"/paper/visqa-x-raying-vision-and-language-reasoning","slug":"visqa-x-raying-vision-and-language-reasoning","title":"VisQA: X-raying Vision and Language Reasoning in Transformers","date":"2021-04-02","arxiv_id":"2104.00926","repositories_listed":1,"syntology":null},{"url":"/paper/an-investigation-of-critical-issues-in-bias","slug":"an-investigation-of-critical-issues-in-bias","title":"Are Bias Mitigation Techniques for Deep Learning Effective?","date":"2021-04-01","arxiv_id":"2104.00170","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/an-investigation-of-critical-issues-in-bias#ran","syntology_url":"https://syntology.ai/paper/2104.00170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00170"}},"official":{"repos":["erobic/bias-mitigators"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/generic-attention-model-explainability-for","slug":"generic-attention-model-explainability-for","title":"Generic Attention-model Explainability for Interpreting Bi-Modal and Encoder-Decoder Transformers","date":"2021-03-29","arxiv_id":"2103.15679","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generic-attention-model-explainability-for#ran","syntology_url":"https://syntology.ai/paper/2103.15679","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.15679"}},"official":{"repos":["hila-chefer/Transformer-MM-Explainability"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/just-because-you-are-right-doesn-t-mean-i-am","slug":"just-because-you-are-right-doesn-t-mean-i-am","title":"'Just because you are right, doesn't mean I am wrong': Overcoming a Bottleneck in the Development and Evaluation of Open-Ended Visual Question Answering (VQA) Tasks","date":"2021-03-28","arxiv_id":"2103.15022","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-answer-validation-for-knowledge","slug":"multi-modal-answer-validation-for-knowledge","title":"Multi-Modal Answer Validation for Knowledge-Based VQA","date":"2021-03-23","arxiv_id":"2103.12248","repositories_listed":1,"syntology":null},{"url":"/paper/select-substitute-search-a-new-benchmark-for","slug":"select-substitute-search-a-new-benchmark-for","title":"Select, Substitute, Search: A New Benchmark for Knowledge-Augmented Visual Question Answering","date":"2021-03-09","arxiv_id":"2103.05568","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-dropout-an-efficient-sample-1","slug":"contextual-dropout-an-efficient-sample-1","title":"Contextual Dropout: An Efficient Sample-Dependent Dropout Module","date":"2021-03-06","arxiv_id":"2103.04181","repositories_listed":1,"syntology":null},{"url":"/paper/visual-question-answering-which-investigated","slug":"visual-question-answering-which-investigated","title":"Visual Question Answering: which investigated applications?","date":"2021-03-04","arxiv_id":"2103.02937","repositories_listed":1,"syntology":null},{"url":"/paper/going-full-tilt-boogie-on-document","slug":"going-full-tilt-boogie-on-document","title":"Going Full-TILT Boogie on Document Understanding with Text-Image-Layout Transformer","date":"2021-02-18","arxiv_id":"2102.09550","repositories_listed":1,"syntology":null},{"url":"/paper/answer-questions-with-right-image-regions-a","slug":"answer-questions-with-right-image-regions-a","title":"Answer Questions with Right Image Regions: A Visual Attention Regularization Approach","date":"2021-02-03","arxiv_id":"2102.01916","repositories_listed":1,"syntology":null},{"url":"/paper/visualmrc-machine-reading-comprehension-on","slug":"visualmrc-machine-reading-comprehension-on","title":"VisualMRC: Machine Reading Comprehension on Document Images","date":"2021-01-27","arxiv_id":"2101.11272","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervision-for-attention-networks","slug":"self-supervision-for-attention-networks","title":"Self Supervision for Attention Networks","date":"2021-01-06","arxiv_id":null,"repositories_listed":1,"syntology":null}],"record_sha256":"87cc85418c1c173c28ffa6ccc302d5c867b157de5522eb39de95561693e8d527","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}