{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/9","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":22,"rows_per_page":100,"rows":[801,900],"of":2167,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering","prev":"/task/visual-question-answering/papers/8","next":"/task/visual-question-answering/papers/10","papers":[{"url":"/paper/chipqa-no-reference-video-quality-prediction","slug":"chipqa-no-reference-video-quality-prediction","title":"ChipQA: No-Reference Video Quality Prediction via Space-Time Chips","date":"2021-09-17","arxiv_id":"2109.08726","repositories_listed":1,"syntology":null},{"url":"/paper/image-captioning-for-effective-use-of","slug":"image-captioning-for-effective-use-of","title":"Image Captioning for Effective Use of Language Models in Knowledge-Based Visual Question Answering","date":"2021-09-15","arxiv_id":"2109.08029","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-the-unknown-knowns-turning","slug":"discovering-the-unknown-knowns-turning","title":"Discovering the Unknown Knowns: Turning Implicit Knowledge in the Dataset into Explicit Training Examples for Visual Question Answering","date":"2021-09-13","arxiv_id":"2109.06122","repositories_listed":1,"syntology":null},{"url":"/paper/xgqa-cross-lingual-visual-question-answering","slug":"xgqa-cross-lingual-visual-question-answering","title":"xGQA: Cross-Lingual Visual Question Answering","date":"2021-09-13","arxiv_id":"2109.06082","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-gpt-3-for-few-shot","slug":"an-empirical-study-of-gpt-3-for-few-shot","title":"An Empirical Study of GPT-3 for Few-Shot Knowledge-Based VQA","date":"2021-09-10","arxiv_id":"2109.05014","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-empirical-study-of-gpt-3-for-few-shot#ran","syntology_url":"https://syntology.ai/paper/2109.05014","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.05014"}},"official":{"repos":["microsoft/PICa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/geneannotator-a-semi-automatic-annotation","slug":"geneannotator-a-semi-automatic-annotation","title":"GeneAnnotator: A Semi-automatic Annotation Tool for Visual Scene Graph","date":"2021-09-06","arxiv_id":"2109.02226","repositories_listed":1,"syntology":null},{"url":"/paper/improved-ramen-towards-domain-generalization","slug":"improved-ramen-towards-domain-generalization","title":"Improved RAMEN: Towards Domain Generalization for Visual Question Answering","date":"2021-09-06","arxiv_id":"2109.02370","repositories_listed":1,"syntology":null},{"url":"/paper/qace-asking-questions-to-evaluate-an-image","slug":"qace-asking-questions-to-evaluate-an-image","title":"QACE: Asking Questions to Evaluate an Image Caption","date":"2021-08-28","arxiv_id":"2108.12560","repositories_listed":1,"syntology":null},{"url":"/paper/task-oriented-multi-user-semantic","slug":"task-oriented-multi-user-semantic","title":"Task-Oriented Multi-User Semantic Communications for VQA Task","date":"2021-08-16","arxiv_id":"2108.07357","repositories_listed":1,"syntology":null},{"url":"/paper/berthop-an-effective-vision-and-language","slug":"berthop-an-effective-vision-and-language","title":"BERTHop: An Effective Vision-and-Language Model for Chest X-ray Disease Diagnosis","date":"2021-08-10","arxiv_id":"2108.04938","repositories_listed":1,"syntology":null},{"url":"/paper/sparse-continuous-distributions-and-fenchel","slug":"sparse-continuous-distributions-and-fenchel","title":"Sparse Continuous Distributions and Fenchel-Young Losses","date":"2021-08-04","arxiv_id":"2108.01988","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sparse-continuous-distributions-and-fenchel#ran","syntology_url":"https://syntology.ai/paper/2108.01988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.01988"}},"official":{"repos":["deep-spin/sparse_continuous_distributions"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/check-it-again-progressive-visual-question-1","slug":"check-it-again-progressive-visual-question-1","title":"Check It Again:Progressive Visual Question Answering via Visual Entailment","date":"2021-08-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/greedy-gradient-ensemble-for-robust-visual","slug":"greedy-gradient-ensemble-for-robust-visual","title":"Greedy Gradient Ensemble for Robust Visual Question Answering","date":"2021-07-27","arxiv_id":"2107.12651","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/greedy-gradient-ensemble-for-robust-visual#ran","syntology_url":"https://syntology.ai/paper/2107.12651","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.12651"}},"official":{"repos":["GeraldHan/GGE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/x-ggm-graph-generative-modeling-for-out-of","slug":"x-ggm-graph-generative-modeling-for-out-of","title":"X-GGM: Graph Generative Modeling for Out-of-Distribution Generalization in Visual Question Answering","date":"2021-07-24","arxiv_id":"2107.11576","repositories_listed":1,"syntology":null},{"url":"/paper/graphhopper-multi-hop-scene-graph-reasoning","slug":"graphhopper-multi-hop-scene-graph-reasoning","title":"Graphhopper: Multi-Hop Scene Graph Reasoning for Visual Question Answering","date":"2021-07-13","arxiv_id":"2107.06325","repositories_listed":1,"syntology":null},{"url":"/paper/dualvgr-a-dual-visual-graph-reasoning-unit","slug":"dualvgr-a-dual-visual-graph-reasoning-unit","title":"DualVGR: A Dual-Visual Graph Reasoning Unit for Video Question Answering","date":"2021-07-10","arxiv_id":"2107.04768","repositories_listed":1,"syntology":null},{"url":"/paper/mind-your-outliers-investigating-the-negative","slug":"mind-your-outliers-investigating-the-negative","title":"Mind Your Outliers! Investigating the Negative Impact of Outliers on Active Learning for Visual Question Answering","date":"2021-07-06","arxiv_id":"2107.02331","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mind-your-outliers-investigating-the-negative#ran","syntology_url":"https://syntology.ai/paper/2107.02331","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2107.02331"}},"official":{"repos":["siddk/vqa-outliers"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cognitive-visual-commonsense-reasoning-using","slug":"cognitive-visual-commonsense-reasoning-using","title":"Cognitive Visual Commonsense Reasoning Using Dynamic Working Memory","date":"2021-07-04","arxiv_id":"2107.01671","repositories_listed":1,"syntology":null},{"url":"/paper/next-qa-next-phase-of-question-answering-to-1","slug":"next-qa-next-phase-of-question-answering-to-1","title":"NExT-QA: Next Phase of Question-Answering to Explaining Temporal Actions","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/perception-matters-detecting-perception","slug":"perception-matters-detecting-perception","title":"Perception Matters: Detecting Perception Failures of VQA Models Using Metamorphic Testing","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/predicting-human-scanpaths-in-visual-question","slug":"predicting-human-scanpaths-in-visual-question","title":"Predicting Human Scanpaths in Visual Question Answering","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rstnet-captioning-with-adaptive-attention-on","slug":"rstnet-captioning-with-adaptive-attention-on","title":"RSTNet: Captioning With Adaptive Attention on Visual and Non-Visual Words","date":"2021-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/probing-image-language-transformers-for-verb","slug":"probing-image-language-transformers-for-verb","title":"Probing Image-Language Transformers for Verb Understanding","date":"2021-06-16","arxiv_id":"2106.09141","repositories_listed":1,"syntology":null},{"url":"/paper/how-modular-should-neural-module-networks-be","slug":"how-modular-should-neural-module-networks-be","title":"How Modular Should Neural Module Networks Be for Systematic Generalization?","date":"2021-06-15","arxiv_id":"2106.08170","repositories_listed":1,"syntology":null},{"url":"/paper/naaqa-a-neural-architecture-for-acoustic","slug":"naaqa-a-neural-architecture-for-acoustic","title":"NAAQA: A Neural Architecture for Acoustic Question Answering","date":"2021-06-11","arxiv_id":"2106.06147","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/naaqa-a-neural-architecture-for-acoustic#ran","syntology_url":"https://syntology.ai/paper/2106.06147","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06147"}},"official":null}},{"url":"/paper/check-it-again-progressive-visual-question","slug":"check-it-again-progressive-visual-question","title":"Check It Again: Progressive Visual Question Answering via Visual Entailment","date":"2021-06-08","arxiv_id":"2106.04605","repositories_listed":1,"syntology":null},{"url":"/paper/clevr-hyp-a-challenge-dataset-and-baselines-1","slug":"clevr-hyp-a-challenge-dataset-and-baselines-1","title":"CLEVR\\_HYP: A Challenge Dataset and Baselines for Visual Question Answering with Hypothetical Actions over Images","date":"2021-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/ease-a-diagnostic-tool-for-vqa-based-on","slug":"ease-a-diagnostic-tool-for-vqa-based-on","title":"EaSe: A Diagnostic Tool for VQA based on Answer Diversity","date":"2021-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/lpf-a-language-prior-feedback-objective","slug":"lpf-a-language-prior-feedback-objective","title":"LPF: A Language-Prior Feedback Objective Function for De-biased Visual Question Answering","date":"2021-05-29","arxiv_id":"2105.14300","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-understanding-and-generation-for","slug":"multi-modal-understanding-and-generation-for","title":"Multi-modal Understanding and Generation for Medical Images and Text via Vision-Language Pre-Training","date":"2021-05-24","arxiv_id":"2105.11333","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-modal-understanding-and-generation-for#ran","syntology_url":"https://syntology.ai/paper/2105.11333","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.11333"}},"official":{"repos":["SuperSupermoon/MedViLL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/structurallm-structural-pre-training-for-form","slug":"structurallm-structural-pre-training-for-form","title":"StructuralLM: Structural Pre-training for Form Understanding","date":"2021-05-24","arxiv_id":"2105.11210","repositories_listed":1,"syntology":{"n":5,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 5 unverified","sample_list":"/paper/structurallm-structural-pre-training-for-form#ran","syntology_url":"https://syntology.ai/paper/2105.11210","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.11210"}},"official":{"repos":["alibaba/AliceMind"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":5,"ran_from_kinds":[]}}},{"url":"/paper/found-a-reason-for-me-weakly-supervised","slug":"found-a-reason-for-me-weakly-supervised","title":"Found a Reason for me? Weakly-supervised Grounded Visual Question Answering using Capsules","date":"2021-05-11","arxiv_id":"2105.04836","repositories_listed":1,"syntology":null},{"url":"/paper/inter-gps-interpretable-geometry-problem","slug":"inter-gps-interpretable-geometry-problem","title":"Inter-GPS: Interpretable Geometry Problem Solving with Formal Language and Symbolic Reasoning","date":"2021-05-10","arxiv_id":"2105.04165","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/inter-gps-interpretable-geometry-problem#ran","syntology_url":"https://syntology.ai/paper/2105.04165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.04165"}},"official":{"repos":["lupantech/InterGPS"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/passage-retrieval-for-outside-knowledge","slug":"passage-retrieval-for-outside-knowledge","title":"Passage Retrieval for Outside-Knowledge Visual Question Answering","date":"2021-05-09","arxiv_id":"2105.03938","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/passage-retrieval-for-outside-knowledge#ran","syntology_url":"https://syntology.ai/paper/2105.03938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2105.03938"}},"official":{"repos":["prdwb/okvqa-release"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adavqa-overcoming-language-priors-with","slug":"adavqa-overcoming-language-priors-with","title":"AdaVQA: Overcoming Language Priors with Adapted Margin Cosine Loss","date":"2021-05-05","arxiv_id":"2105.01993","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-training-of-variational-quantum","slug":"optimal-training-of-variational-quantum","title":"Optimal training of variational quantum algorithms without barren plateaus","date":"2021-04-29","arxiv_id":"2104.14543","repositories_listed":1,"syntology":null},{"url":"/paper/reltransformer-balancing-the-visual","slug":"reltransformer-balancing-the-visual","title":"RelTransformer: A Transformer-Based Long-Tail Visual Relationship Recognition","date":"2021-04-24","arxiv_id":"2104.11934","repositories_listed":1,"syntology":null},{"url":"/paper/graghvqa-language-guided-graph-neural","slug":"graghvqa-language-guided-graph-neural","title":"GraghVQA: Language-Guided Graph Neural Networks for Graph-based Visual Question Answering","date":"2021-04-20","arxiv_id":"2104.10283","repositories_listed":1,"syntology":null},{"url":"/paper/clevr-hyp-a-challenge-dataset-and-baselines","slug":"clevr-hyp-a-challenge-dataset-and-baselines","title":"CLEVR_HYP: A Challenge Dataset and Baselines for Visual Question Answering with Hypothetical Actions over Images","date":"2021-04-13","arxiv_id":"2104.05981","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-question-based-biases-assessing","slug":"beyond-question-based-biases-assessing","title":"Beyond Question-Based Biases: Assessing Multimodal Shortcut Learning in Visual Question Answering","date":"2021-04-07","arxiv_id":"2104.03149","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-question-based-biases-assessing#ran","syntology_url":"https://syntology.ai/paper/2104.03149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.03149"}},"official":{"repos":["cdancette/detect-shortcuts"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mmbert-multimodal-bert-pretraining-for","slug":"mmbert-multimodal-bert-pretraining-for","title":"MMBERT: Multimodal BERT Pretraining for Improved Medical VQA","date":"2021-04-03","arxiv_id":"2104.01394","repositories_listed":1,"syntology":null},{"url":"/paper/visqa-x-raying-vision-and-language-reasoning","slug":"visqa-x-raying-vision-and-language-reasoning","title":"VisQA: X-raying Vision and Language Reasoning in Transformers","date":"2021-04-02","arxiv_id":"2104.00926","repositories_listed":1,"syntology":null},{"url":"/paper/an-investigation-of-critical-issues-in-bias","slug":"an-investigation-of-critical-issues-in-bias","title":"Are Bias Mitigation Techniques for Deep Learning Effective?","date":"2021-04-01","arxiv_id":"2104.00170","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/an-investigation-of-critical-issues-in-bias#ran","syntology_url":"https://syntology.ai/paper/2104.00170","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2104.00170"}},"official":{"repos":["erobic/bias-mitigators"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/generic-attention-model-explainability-for","slug":"generic-attention-model-explainability-for","title":"Generic Attention-model Explainability for Interpreting Bi-Modal and Encoder-Decoder Transformers","date":"2021-03-29","arxiv_id":"2103.15679","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generic-attention-model-explainability-for#ran","syntology_url":"https://syntology.ai/paper/2103.15679","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2103.15679"}},"official":{"repos":["hila-chefer/Transformer-MM-Explainability"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/just-because-you-are-right-doesn-t-mean-i-am","slug":"just-because-you-are-right-doesn-t-mean-i-am","title":"'Just because you are right, doesn't mean I am wrong': Overcoming a Bottleneck in the Development and Evaluation of Open-Ended Visual Question Answering (VQA) Tasks","date":"2021-03-28","arxiv_id":"2103.15022","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-hidden-treasure-of-dialog-in-video","slug":"on-the-hidden-treasure-of-dialog-in-video","title":"On the hidden treasure of dialog in video question answering","date":"2021-03-26","arxiv_id":"2103.14517","repositories_listed":1,"syntology":null},{"url":"/paper/multi-modal-answer-validation-for-knowledge","slug":"multi-modal-answer-validation-for-knowledge","title":"Multi-Modal Answer Validation for Knowledge-Based VQA","date":"2021-03-23","arxiv_id":"2103.12248","repositories_listed":1,"syntology":null},{"url":"/paper/select-substitute-search-a-new-benchmark-for","slug":"select-substitute-search-a-new-benchmark-for","title":"Select, Substitute, Search: A New Benchmark for Knowledge-Augmented Visual Question Answering","date":"2021-03-09","arxiv_id":"2103.05568","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-dropout-an-efficient-sample-1","slug":"contextual-dropout-an-efficient-sample-1","title":"Contextual Dropout: An Efficient Sample-Dependent Dropout Module","date":"2021-03-06","arxiv_id":"2103.04181","repositories_listed":1,"syntology":null},{"url":"/paper/visual-question-answering-which-investigated","slug":"visual-question-answering-which-investigated","title":"Visual Question Answering: which investigated applications?","date":"2021-03-04","arxiv_id":"2103.02937","repositories_listed":1,"syntology":null},{"url":"/paper/going-full-tilt-boogie-on-document","slug":"going-full-tilt-boogie-on-document","title":"Going Full-TILT Boogie on Document Understanding with Text-Image-Layout Transformer","date":"2021-02-18","arxiv_id":"2102.09550","repositories_listed":1,"syntology":null},{"url":"/paper/less-is-more-clipbert-for-video-and-language","slug":"less-is-more-clipbert-for-video-and-language","title":"Less is More: ClipBERT for Video-and-Language Learning via Sparse Sampling","date":"2021-02-11","arxiv_id":"2102.06183","repositories_listed":1,"syntology":null},{"url":"/paper/answer-questions-with-right-image-regions-a","slug":"answer-questions-with-right-image-regions-a","title":"Answer Questions with Right Image Regions: A Visual Attention Regularization Approach","date":"2021-02-03","arxiv_id":"2102.01916","repositories_listed":1,"syntology":null},{"url":"/paper/visualmrc-machine-reading-comprehension-on","slug":"visualmrc-machine-reading-comprehension-on","title":"VisualMRC: Machine Reading Comprehension on Document Images","date":"2021-01-27","arxiv_id":"2101.11272","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervision-for-attention-networks","slug":"self-supervision-for-attention-networks","title":"Self Supervision for Attention Networks","date":"2021-01-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/mdetr-modulated-detection-for-end-to-end-1","slug":"mdetr-modulated-detection-for-end-to-end-1","title":"MDETR - Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-co-attention-transformer-for","slug":"multimodal-co-attention-transformer-for","title":"Multimodal Co-Attention Transformer for Survival Prediction in Gigapixel Whole Slide Images","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/pano-avqa-grounded-audio-visual-question","slug":"pano-avqa-grounded-audio-visual-question","title":"Pano-AVQA: Grounded Audio-Visual Question Answering on 360deg Videos","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/trar-routing-the-attention-spans-in","slug":"trar-routing-the-attention-spans-in","title":"TRAR: Routing the Attention Spans in Transformer for Visual Question Answering","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-curriculum-domain-adaptation-for","slug":"unsupervised-curriculum-domain-adaptation-for","title":"Unsupervised Curriculum Domain Adaptation for No-Reference Video Quality Assessment","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/detecting-hate-speech-in-multi-modal-memes","slug":"detecting-hate-speech-in-multi-modal-memes","title":"Detecting Hate Speech in Multi-modal Memes","date":"2020-12-29","arxiv_id":"2012.14891","repositories_listed":1,"syntology":null},{"url":"/paper/learning-content-and-context-with-language","slug":"learning-content-and-context-with-language","title":"Learning content and context with language bias for Visual Question Answering","date":"2020-12-21","arxiv_id":"2012.11134","repositories_listed":1,"syntology":null},{"url":"/paper/on-modality-bias-in-the-tvqa-dataset","slug":"on-modality-bias-in-the-tvqa-dataset","title":"On Modality Bias in the TVQA Dataset","date":"2020-12-18","arxiv_id":"2012.10210","repositories_listed":1,"syntology":null},{"url":"/paper/overcoming-language-priors-with-self","slug":"overcoming-language-priors-with-self","title":"Overcoming Language Priors with Self-supervised Learning for Visual Question Answering","date":"2020-12-17","arxiv_id":"2012.11528","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":3,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/overcoming-language-priors-with-self#ran","syntology_url":"https://syntology.ai/paper/2012.11528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.11528"}},"official":{"repos":["CrossmodalGroup/SSL-VQA"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/knowledge-routed-visual-question-reasoning","slug":"knowledge-routed-visual-question-reasoning","title":"Knowledge-Routed Visual Question Reasoning: Challenges for Deep Representation Embedding","date":"2020-12-14","arxiv_id":"2012.07192","repositories_listed":1,"syntology":null},{"url":"/paper/simple-is-not-easy-a-simple-strong-baseline","slug":"simple-is-not-easy-a-simple-strong-baseline","title":"Simple is not Easy: A Simple Strong Baseline for TextVQA and TextCaps","date":"2020-12-09","arxiv_id":"2012.05153","repositories_listed":1,"syntology":null},{"url":"/paper/craft-a-benchmark-for-causal-reasoning-about","slug":"craft-a-benchmark-for-causal-reasoning-about","title":"CRAFT: A Benchmark for Causal Reasoning About Forces and inTeractions","date":"2020-12-08","arxiv_id":"2012.04293","repositories_listed":1,"syntology":null},{"url":"/paper/tap-text-aware-pre-training-for-text-vqa-and","slug":"tap-text-aware-pre-training-for-text-vqa-and","title":"TAP: Text-Aware Pre-training for Text-VQA and Text-Caption","date":"2020-12-08","arxiv_id":"2012.04638","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-guided-image-captioning","slug":"understanding-guided-image-captioning","title":"Understanding Guided Image Captioning Performance across Domains","date":"2020-12-04","arxiv_id":"2012.02339","repositories_listed":1,"syntology":null},{"url":"/paper/just-ask-learning-to-answer-questions-from","slug":"just-ask-learning-to-answer-questions-from","title":"Just Ask: Learning to Answer Questions from Millions of Narrated Videos","date":"2020-12-01","arxiv_id":"2012.00451","repositories_listed":1,"syntology":null},{"url":"/paper/open-ended-multi-modal-relational-reason-for","slug":"open-ended-multi-modal-relational-reason-for","title":"Open-Ended Multi-Modal Relational Reasoning for Video Question Answering","date":"2020-12-01","arxiv_id":"2012.00822","repositories_listed":1,"syntology":null},{"url":"/paper/towards-knowledge-augmented-visual-question","slug":"towards-knowledge-augmented-visual-question","title":"Towards Knowledge-Augmented Visual Question Answering","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/patch-vq-patching-up-the-video-quality","slug":"patch-vq-patching-up-the-video-quality","title":"Patch-VQ: 'Patching Up' the Video Quality Problem","date":"2020-11-27","arxiv_id":"2011.13544","repositories_listed":1,"syntology":null},{"url":"/paper/point-and-ask-incorporating-pointing-into","slug":"point-and-ask-incorporating-pointing-into","title":"Point and Ask: Incorporating Pointing into Visual Question Answering","date":"2020-11-27","arxiv_id":"2011.13681","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-lexical-perturbations-for","slug":"learning-from-lexical-perturbations-for","title":"Learning from Lexical Perturbations for Consistent Visual Question Answering","date":"2020-11-26","arxiv_id":"2011.13406","repositories_listed":1,"syntology":null},{"url":"/paper/transformation-driven-visual-reasoning","slug":"transformation-driven-visual-reasoning","title":"Transformation Driven Visual Reasoning","date":"2020-11-26","arxiv_id":"2011.13160","repositories_listed":1,"syntology":null},{"url":"/paper/interpretable-visual-reasoning-via-induced","slug":"interpretable-visual-reasoning-via-induced","title":"Interpretable Visual Reasoning via Induced Symbolic Space","date":"2020-11-23","arxiv_id":"2011.11603","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-multimodal-classification-using","slug":"large-scale-multimodal-classification-using","title":"Large Scale Multimodal Classification Using an Ensemble of Transformer Models and Co-Attention","date":"2020-11-23","arxiv_id":"2011.11735","repositories_listed":1,"syntology":null},{"url":"/paper/siamese-tracking-with-lingual-object","slug":"siamese-tracking-with-lingual-object","title":"Siamese Tracking with Lingual Object Constraints","date":"2020-11-23","arxiv_id":"2011.11721","repositories_listed":1,"syntology":null},{"url":"/paper/capwap-captioning-with-a-purpose","slug":"capwap-captioning-with-a-purpose","title":"CapWAP: Captioning with a Purpose","date":"2020-11-09","arxiv_id":"2011.04264","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-model-and-ignore-dataset-bias","slug":"learning-to-model-and-ignore-dataset-bias","title":"Learning to Model and Ignore Dataset Bias with Mixed Capacity Ensembles","date":"2020-11-07","arxiv_id":"2011.03856","repositories_listed":1,"syntology":{"n":18,"n_ran":10,"n_constructed":5,"n_ran_checked":8,"n_instrument":2,"n_unverified":8,"n_honours":2,"n_violates":1,"n_no_contract":5,"n_pointer_only":0,"phrase":"10 ran (of which 5 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/learning-to-model-and-ignore-dataset-bias#ran","syntology_url":"https://syntology.ai/paper/2011.03856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.03856"}},"official":{"repos":["chrisc36/autobias"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":5,"n_ran_no_instrument_failure":8,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/disentangling-3d-prototypical-networks-for-1","slug":"disentangling-3d-prototypical-networks-for-1","title":"Disentangling 3D Prototypical Networks For Few-Shot Concept Learning","date":"2020-11-06","arxiv_id":"2011.03367","repositories_listed":1,"syntology":null},{"url":"/paper/an-improved-attention-for-visual-question","slug":"an-improved-attention-for-visual-question","title":"An Improved Attention for Visual Question Answering","date":"2020-11-04","arxiv_id":"2011.02164","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-contrast-the-counterfactual","slug":"learning-to-contrast-the-counterfactual","title":"Learning to Contrast the Counterfactual Samples for Robust Visual Question Answering","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/loss-rescaling-vqa-revisiting-language-prior","slug":"loss-rescaling-vqa-revisiting-language-prior","title":"Loss re-scaling VQA: Revisiting the LanguagePrior Problem from a Class-imbalance View","date":"2020-10-30","arxiv_id":"2010.16010","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/loss-rescaling-vqa-revisiting-language-prior#ran","syntology_url":"https://syntology.ai/paper/2010.16010","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.16010"}},"official":{"repos":["guoyang9/class-imbalance-VQA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mmft-bert-multimodal-fusion-transformer-with","slug":"mmft-bert-multimodal-fusion-transformer-with","title":"MMFT-BERT: Multimodal Fusion Transformer with BERT Encodings for Visual Question Answering","date":"2020-10-27","arxiv_id":"2010.14095","repositories_listed":1,"syntology":null},{"url":"/paper/st-greed-space-time-generalized-entropic","slug":"st-greed-space-time-generalized-entropic","title":"ST-GREED: Space-Time Generalized Entropic Differences for Frame Rate Dependent Video Quality Prediction","date":"2020-10-26","arxiv_id":"2010.13715","repositories_listed":1,"syntology":null},{"url":"/paper/ruart-a-novel-text-centered-solution-for-text","slug":"ruart-a-novel-text-centered-solution-for-text","title":"RUArt: A Novel Text-Centered Solution for Text-Based Visual Question Answering","date":"2020-10-24","arxiv_id":"2010.12917","repositories_listed":1,"syntology":null},{"url":"/paper/removing-bias-in-multi-modal-classifiers","slug":"removing-bias-in-multi-modal-classifiers","title":"Removing Bias in Multi-modal Classifiers: Regularization by Maximizing Functional Entropies","date":"2020-10-21","arxiv_id":"2010.10802","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/removing-bias-in-multi-modal-classifiers#ran","syntology_url":"https://syntology.ai/paper/2010.10802","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10802"}},"official":{"repos":["itaigat/removing-bias-in-multi-modal-classifiers"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/bayesian-attention-modules","slug":"bayesian-attention-modules","title":"Bayesian Attention Modules","date":"2020-10-20","arxiv_id":"2010.10604","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/bayesian-attention-modules#ran","syntology_url":"https://syntology.ai/paper/2010.10604","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10604"}},"official":{"repos":["zhougroup/BAM"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/sort-ing-vqa-models-contrastive-gradient","slug":"sort-ing-vqa-models-contrastive-gradient","title":"SOrT-ing VQA Models : Contrastive Gradient Learning for Improved Consistency","date":"2020-10-20","arxiv_id":"2010.10038","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sort-ing-vqa-models-contrastive-gradient#ran","syntology_url":"https://syntology.ai/paper/2010.10038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10038"}},"official":{"repos":["sameerdharur/sorting-vqa"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/natural-language-rationales-with-full-stack","slug":"natural-language-rationales-with-full-stack","title":"Natural Language Rationales with Full-Stack Visual Reasoning: From Pixels to Semantic Frames to Commonsense Graphs","date":"2020-10-15","arxiv_id":"2010.07526","repositories_listed":1,"syntology":null},{"url":"/paper/contrast-and-classify-alternate-training-for","slug":"contrast-and-classify-alternate-training-for","title":"Contrast and Classify: Training Robust VQA Models","date":"2020-10-13","arxiv_id":"2010.06087","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-deep-multi-modal-network-for","slug":"hierarchical-deep-multi-modal-network-for","title":"Hierarchical Deep Multi-modal Network for Medical Visual Question Answering","date":"2020-09-27","arxiv_id":"2009.12770","repositories_listed":1,"syntology":null},{"url":"/paper/multiple-interaction-learning-with-question","slug":"multiple-interaction-learning-with-question","title":"Multiple interaction learning with question-type prior knowledge for constraining answer search space in visual question answering","date":"2020-09-23","arxiv_id":"2009.11118","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparison-of-pre-trained-vision-and","slug":"a-comparison-of-pre-trained-vision-and","title":"A Comparison of Pre-trained Vision-and-Language Models for Multimodal Representation Learning across Medical Images and Reports","date":"2020-09-03","arxiv_id":"2009.01523","repositories_listed":1,"syntology":null},{"url":"/paper/a-dataset-and-baselines-for-visual-question","slug":"a-dataset-and-baselines-for-visual-question","title":"A Dataset and Baselines for Visual Question Answering on Art","date":"2020-08-28","arxiv_id":"2008.12520","repositories_listed":1,"syntology":null},{"url":"/paper/no-reference-video-quality-assessment-using-1","slug":"no-reference-video-quality-assessment-using-1","title":"No-Reference Video Quality Assessment Using Space-Time Chips","date":"2020-08-23","arxiv_id":"2008.00031","repositories_listed":1,"syntology":null},{"url":"/paper/devlbert-learning-deconfounded-visio","slug":"devlbert-learning-deconfounded-visio","title":"DeVLBert: Learning Deconfounded Visio-Linguistic Representations","date":"2020-08-16","arxiv_id":"2008.06884","repositories_listed":1,"syntology":null},{"url":"/paper/noise-induced-barren-plateaus-in-variational","slug":"noise-induced-barren-plateaus-in-variational","title":"Noise-Induced Barren Plateaus in Variational Quantum Algorithms","date":"2020-07-28","arxiv_id":"2007.14384","repositories_listed":1,"syntology":null}],"record_sha256":"185825457187b45a5f2ef0d582438574dd29075b2ba5d06c675efc396c303f2b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}