{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/visual-question-answering/papers/17","list_of":"/task/visual-question-answering","task":"Visual Question Answering (VQA)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":17,"pages_in_order":22,"rows_per_page":100,"rows":[1601,1700],"of":2167,"counts":{"archive_papers_tagged":2167,"with_a_code_link":1039,"where_syntology_ran_a_sample":359,"not_listed_spam_title":0,"listed":2167,"listed_where_code_ran":359,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":287,"every_run_a_failure_of_syntologys_instrument":72,"listed_with_a_run_with_no_instrument_failure":287,"listed_every_run_a_failure_of_syntologys_instrument":72,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/visual-question-answering","prev":"/task/visual-question-answering/papers/16","next":"/task/visual-question-answering/papers/18","papers":[{"url":null,"slug":"unicon-unidirectional-split-learning-with","title":"Bidirectional Contrastive Split Learning for Visual Question Answering","date":"2022-08-24","arxiv_id":"2208.11435","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-good-are-deep-models-in-understanding-the","title":"How good are deep models in understanding the generated images?","date":"2022-08-23","arxiv_id":"2208.10760","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlmae-vision-language-masked-autoencoder","title":"VLMAE: Vision-Language Masked Autoencoder","date":"2022-08-19","arxiv_id":"2208.09374","repositories_listed":0,"syntology":null},{"url":"/paper/aesthetic-visual-question-answering-of","slug":"aesthetic-visual-question-answering-of","title":"Aesthetic Visual Question Answering of Photographs","date":"2022-08-10","arxiv_id":"2208.05798","repositories_listed":0,"syntology":null},{"url":null,"slug":"pan-pulse-ansatz-on-nisq-machines","title":"NAPA: Intermediate-level Variational Native-pulse Ansatz for Variational Quantum Algorithms","date":"2022-08-02","arxiv_id":"2208.01215","repositories_listed":0,"syntology":null},{"url":"/paper/video-question-answering-with-iterative-video","slug":"video-question-answering-with-iterative-video","title":"Video Question Answering with Iterative Video-Text Co-Tokenization","date":"2022-08-01","arxiv_id":"2208.00934","repositories_listed":0,"syntology":null},{"url":null,"slug":"parameter-parallel-distributed-variational","title":"Parameter-Parallel Distributed Variational Quantum Algorithm","date":"2022-07-31","arxiv_id":"2208.00450","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-based-visual-question-answering-1","title":"Uncertainty-based Visual Question Answering: Estimating Semantic Inconsistency between Image and Knowledge Base","date":"2022-07-27","arxiv_id":"2207.13242","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-gpt-3-all-you-need-for-visual-question","title":"Is GPT-3 all you need for Visual Question Answering in Cultural Heritage?","date":"2022-07-25","arxiv_id":"2207.12101","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-complex-document-understanding-by","title":"Towards Complex Document Understanding By Discrete Reasoning","date":"2022-07-25","arxiv_id":"2207.11871","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-perturbation-aware-collaborative","title":"Visual Perturbation-aware Collaborative Learning for Overcoming the Language Prior Problem","date":"2022-07-24","arxiv_id":"2207.11850","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-aware-modular-capsule-routing-for","title":"Semantic-aware Modular Capsule Routing for Visual Question Answering","date":"2022-07-21","arxiv_id":"2207.10404","repositories_listed":0,"syntology":null},{"url":null,"slug":"qsan-a-near-term-achievable-quantum-self","title":"QSAN: A Near-term Achievable Quantum Self-Attention Network","date":"2022-07-14","arxiv_id":"2207.07563","repositories_listed":0,"syntology":null},{"url":"/paper/ovqa-a-clinically-generated-visual-question","slug":"ovqa-a-clinically-generated-visual-question","title":"OVQA: A Clinically Generated Visual Question Answering Dataset","date":"2022-07-07","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"american-white-in-multimodal-language-and","title":"American == White in Multimodal Language-and-Image AI","date":"2022-07-01","arxiv_id":"2207.00691","repositories_listed":0,"syntology":null},{"url":null,"slug":"vgnmn-video-grounded-neural-module-networks","title":"VGNMN: Video-grounded Neural Module Networks for Video-Grounded Dialogue Systems","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"modern-question-answering-datasets-and","title":"Modern Question Answering Datasets and Benchmarks: A Survey","date":"2022-06-30","arxiv_id":"2206.15030","repositories_listed":0,"syntology":null},{"url":null,"slug":"ebms-vs-cl-exploring-self-supervised-visual","title":"EBMs vs. CL: Exploring Self-Supervised Visual Pretraining for Visual Question Answering","date":"2022-06-29","arxiv_id":"2206.14355","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-shallow-to-deep-compositional-reasoning","title":"From Shallow to Deep: Compositional Reasoning over Graphs for Visual Question Answering","date":"2022-06-25","arxiv_id":"2206.12533","repositories_listed":0,"syntology":null},{"url":null,"slug":"tell-me-the-evidence-dual-visual-linguistic","title":"Tell Me the Evidence? Dual Visual-Linguistic Interaction for Answer Grounding","date":"2022-06-21","arxiv_id":"2207.05703","repositories_listed":0,"syntology":null},{"url":null,"slug":"grounding-answers-for-visual-questions-asked-1","title":"Grounding Answers for Visual Questions Asked by Visually Impaired People","date":"2022-06-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/unified-io-a-unified-model-for-vision","slug":"unified-io-a-unified-model-for-vision","title":"Unified-IO: A Unified Model for Vision, Language, and Multi-Modal Tasks","date":"2022-06-17","arxiv_id":"2206.08916","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-time-adaptation-for-visual-document","title":"Test-Time Adaptation for Visual Document Understanding","date":"2022-06-15","arxiv_id":"2206.07240","repositories_listed":0,"syntology":null},{"url":"/paper/less-is-more-linear-layers-on-clip-features","slug":"less-is-more-linear-layers-on-clip-features","title":"Less Is More: Linear Layers on CLIP Features as Powerful VizWiz Model","date":"2022-06-10","arxiv_id":"2206.05281","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-pixels-to-objects-cubic-visual-attention","title":"From Pixels to Objects: Cubic Visual Attention for Visual Question Answering","date":"2022-06-04","arxiv_id":"2206.01923","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-two-stream-attention-network-for","title":"Structured Two-stream Attention Network for Video Question Answering","date":"2022-06-02","arxiv_id":"2206.01017","repositories_listed":0,"syntology":null},{"url":null,"slug":"vl-beit-generative-vision-language","title":"VL-BEiT: Generative Vision-Language Pretraining","date":"2022-06-02","arxiv_id":"2206.01127","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-vs-from-scratch-do-vision","title":"Fine-tuning vs From Scratch: Do Vision & Language Models Have Similar Capabilities on Out-of-Distribution Visual Question Answering?","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"question-modifiers-in-visual-question","title":"Question Modifiers in Visual Question Answering","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"un-jeu-de-donnees-pour-repondre-a-des","title":"Un jeu de données pour répondre à des questions visuelles à propos d’entités nommées en utilisant des bases de connaissances (ViQuAE, a Dataset for Knowledge-based Visual Question Answering about Named Entities)","date":"2022-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-superordinate-abstraction-for-robust","title":"Visual Superordinate Abstraction for Robust Concept Learning","date":"2022-05-28","arxiv_id":"2205.14444","repositories_listed":0,"syntology":null},{"url":null,"slug":"v-doc-visual-questions-answers-with-documents","title":"V-Doc : Visual questions answers with Documents","date":"2022-05-27","arxiv_id":"2205.13724","repositories_listed":0,"syntology":null},{"url":null,"slug":"avoiding-barren-plateaus-with-classical-deep","title":"Avoiding Barren Plateaus with Classical Deep Neural Networks","date":"2022-05-26","arxiv_id":"2205.13418","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-visual-question-answering-with","title":"Guiding Visual Question Answering with Attention Priors","date":"2022-05-25","arxiv_id":"2205.12616","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-advances-in-text-generation-from-images","title":"On Advances in Text Generation from Images Beyond Captioning: A Case Study in Self-Rationalization","date":"2022-05-24","arxiv_id":"2205.11686","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-evaluation-practices-in-visual","title":"Reassessing Evaluation Practices in Visual Question Answering: A Case Study on Out-of-Distribution Generalization","date":"2022-05-24","arxiv_id":"2205.12191","repositories_listed":0,"syntology":null},{"url":null,"slug":"vqa-gnn-reasoning-with-multimodal-semantic","title":"VQA-GNN: Reasoning with Multimodal Knowledge via Graph Neural Networks for Visual Question Answering","date":"2022-05-23","arxiv_id":"2205.11501","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-video-quality-assessment-models","title":"Making Video Quality Assessment Models Sensitive to Frame Rate Distortions","date":"2022-05-21","arxiv_id":"2205.10501","repositories_listed":0,"syntology":null},{"url":null,"slug":"gender-and-racial-bias-in-visual-question","title":"Gender and Racial Bias in Visual Question Answering Datasets","date":"2022-05-17","arxiv_id":"2205.08148","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-framework-to-map-vmaf-with-the-probability","title":"A Framework to Map VMAF with the Probability of Just Noticeable Difference between Video Encoding Recipes","date":"2022-05-16","arxiv_id":"2205.07565","repositories_listed":0,"syntology":null},{"url":null,"slug":"serving-and-optimizing-machine-learning","title":"Serving and Optimizing Machine Learning Workflows on Heterogeneous Infrastructures","date":"2022-05-10","arxiv_id":"2205.04713","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-learning-of-object-graph-and-relation","title":"Joint learning of object graph and relation graph for visual question answering","date":"2022-05-09","arxiv_id":"2205.04188","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-quality-assessment-of-compressed-videos","title":"Deep Quality Assessment of Compressed Videos: A Subjective and Objective Study","date":"2022-05-07","arxiv_id":"2205.03630","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-easy-to-hard-learning-language-guided","title":"From Easy to Hard: Learning Language-guided Curriculum for Visual Question Answering on Remote Sensing Data","date":"2022-05-06","arxiv_id":"2205.03147","repositories_listed":0,"syntology":null},{"url":null,"slug":"answer-me-multi-task-open-vocabulary-visual","title":"Answer-Me: Multi-Task Open-Vocabulary Visual Question Answering","date":"2022-05-02","arxiv_id":"2205.00949","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-recognition-level","title":"Bridging the Gap between Recognition-level Pre-training and Commonsensical Vision-language Tasks","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-language-pretraining-current-trends","title":"Vision-Language Pretraining: Current Trends and the Future","date":"2022-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/multimodal-adaptive-distillation-for","slug":"multimodal-adaptive-distillation-for","title":"Multimodal Adaptive Distillation for Leveraging Unimodal Encoders for Vision-Language Tasks","date":"2022-04-22","arxiv_id":"2204.10496","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-mechanism-based-cognition-level","title":"Attention Mechanism based Cognition-level Scene Understanding","date":"2022-04-17","arxiv_id":"2204.08027","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-cross-modal-understanding-in-visual","title":"Improving Cross-Modal Understanding in Visual Dialog via Contrastive Learning","date":"2022-04-15","arxiv_id":"2204.07302","repositories_listed":0,"syntology":null},{"url":null,"slug":"question-driven-graph-fusion-network-for","title":"Question-Driven Graph Fusion Network For Visual Question Answering","date":"2022-04-03","arxiv_id":"2204.00975","repositories_listed":0,"syntology":null},{"url":null,"slug":"co-vqa-answering-by-interactive-sub-question-1","title":"Co-VQA : Answering by Interactive Sub Question Sequence","date":"2022-04-02","arxiv_id":"2204.00879","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceptual-quality-assessment-of-ugc-gaming","title":"Perceptual Quality Assessment of UGC Gaming Videos","date":"2022-03-31","arxiv_id":"2204.00128","repositories_listed":0,"syntology":null},{"url":null,"slug":"simvqa-exploring-simulated-environments-for","title":"SimVQA: Exploring Simulated Environments for Visual Question Answering","date":"2022-03-31","arxiv_id":"2203.17219","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-mechanisms-inspired-efficient","title":"Visual Mechanisms Inspired Efficient Transformers for Image and Video Quality Assessment","date":"2022-03-28","arxiv_id":"2203.14557","repositories_listed":0,"syntology":null},{"url":null,"slug":"subjective-and-objective-analysis-of-streamed","title":"Subjective and Objective Analysis of Streamed Gaming Videos","date":"2022-03-24","arxiv_id":"2203.12824","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-escaping-from-language-bias-and-ocr","title":"Towards Escaping from Language Bias and OCR Error: Semantics-Centered Text Visual Question Answering","date":"2022-03-24","arxiv_id":"2203.12929","repositories_listed":0,"syntology":null},{"url":null,"slug":"wudaomm-a-large-scale-multi-modal-dataset-for","title":"WuDaoMM: A large-scale Multi-Modal Dataset for Pre-training models","date":"2022-03-22","arxiv_id":"2203.11480","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-you-even-tell-left-from-right-presenting","title":"Can you even tell left from right? Presenting a new challenge for VQA","date":"2022-03-15","arxiv_id":"2203.07664","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-models-are-few-shot-learners-empirical","title":"CLIP Models are Few-shot Learners: Empirical Studies on VQA and Visual Entailment","date":"2022-03-14","arxiv_id":"2203.07190","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-multimodal-generation-on-clip-via-1","title":"Enabling Multimodal Generation on CLIP via Vision-Language Knowledge Distillation","date":"2022-03-12","arxiv_id":"2203.06386","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-coreference-relations-in-visual-1","title":"Modeling Coreference Relations in Visual Dialog","date":"2022-03-06","arxiv_id":"2203.02986","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-rapid-advancement-in-visual-question","title":"Recent, rapid advancement in visual question answering architecture: a review","date":"2022-03-02","arxiv_id":"2203.01322","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-vision-and-language-pre-training","title":"Unsupervised Vision-and-Language Pre-training via Retrieval-based Multi-Granular Alignment","date":"2022-03-01","arxiv_id":"2203.00242","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-clevrness-blackbox-testing-of","title":"Measuring CLEVRness: Blackbox testing of Visual Reasoning Models","date":"2022-02-24","arxiv_id":"2202.12162","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-vqa-based-on-a-novel-hybrid-training","title":"RankDVQA: Deep VQA based on Ranking-inspired Hybrid Training","date":"2022-02-17","arxiv_id":"2202.08595","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-visual-question-answering","title":"Privacy Preserving Visual Question Answering","date":"2022-02-15","arxiv_id":"2202.07712","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-experimental-study-of-the-vision","title":"An experimental study of the vision-bottleneck in VQA","date":"2022-02-14","arxiv_id":"2202.06858","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-open-domain-question-answering-systems","title":"Can Open Domain Question Answering Systems Answer Visual Knowledge Questions?","date":"2022-02-09","arxiv_id":"2202.04306","repositories_listed":0,"syntology":null},{"url":"/paper/newskvqa-knowledge-aware-news-video-question","slug":"newskvqa-knowledge-aware-news-video-question","title":"NEWSKVQA: Knowledge-Aware News Video Question Answering","date":"2022-02-08","arxiv_id":"2202.04015","repositories_listed":0,"syntology":null},{"url":"/paper/webly-supervised-concept-expansion-for","slug":"webly-supervised-concept-expansion-for","title":"Webly Supervised Concept Expansion for General Purpose Vision Models","date":"2022-02-04","arxiv_id":"2202.02317","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-compose-diversified-prompts-for","title":"Learning to Compose Diversified Prompts for Image Emotion Classification","date":"2022-01-26","arxiv_id":"2201.10963","repositories_listed":0,"syntology":null},{"url":null,"slug":"mga-vqa-multi-granularity-alignment-for-1","title":"MGA-VQA: Multi-Granularity Alignment for Visual Question Answering","date":"2022-01-25","arxiv_id":"2201.10656","repositories_listed":0,"syntology":null},{"url":null,"slug":"sa-vqa-structured-alignment-of-visual-and","title":"SA-VQA: Structured Alignment of Visual and Semantic Representations for Visual Question Answering","date":"2022-01-25","arxiv_id":"2201.10654","repositories_listed":0,"syntology":null},{"url":null,"slug":"question-generation-for-evaluating-cross","title":"Question Generation for Evaluating Cross-Dataset Shifts in Multi-modal Grounding","date":"2022-01-24","arxiv_id":"2201.09639","repositories_listed":0,"syntology":null},{"url":null,"slug":"all-you-may-need-for-vqa-are-image-captions","title":"All You May Need for VQA are Image Captions","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"kat-a-knowledge-augmented-transformer-for-1","title":"KAT: A Knowledge Augmented Transformer for Vision-and-Language","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mango-enhancing-the-robustness-of-vqa-models","title":"MANGO: Enhancing the Robustness of VQA Models via Adversarial Noise Generation","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-the-role-of-positional-information-in","title":"Probing the Role of Positional Information in Vision-Language Models","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"retrieving-visual-facts-for-few-shot-visual","title":"Retrieving Visual Facts For Few-Shot Visual Question Answering","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"task-formulation-matters-when-learning","title":"Task Formulation Matters When Learning Continuously: A Case Study in Visual Question Answering","date":"2022-01-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-td-clip-targeted-distillation-for-vision","title":"CLIP-TD: CLIP Targeted Distillation for Vision-Language Tasks","date":"2022-01-15","arxiv_id":"2201.05729","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-thousand-words-are-worth-more-than-a","title":"A Thousand Words Are Worth More Than a Picture: Natural Language-Centric Outside-Knowledge Visual Question Answering","date":"2022-01-14","arxiv_id":"2201.05299","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-automated-error-analysis-learning-to","title":"Towards Automated Error Analysis: Learning to Characterize Errors","date":"2022-01-13","arxiv_id":"2201.05017","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-efficacy-of-co-attention-transformer","title":"On the Efficacy of Co-Attention Transformer Layers in Visual Question Answering","date":"2022-01-11","arxiv_id":"2201.03965","repositories_listed":0,"syntology":null},{"url":null,"slug":"uni-eden-universal-encoder-decoder-network-by","title":"Uni-EDEN: Universal Encoder-Decoder Network by Multi-Granular Vision-Language Pre-training","date":"2022-01-11","arxiv_id":"2201.04026","repositories_listed":0,"syntology":null},{"url":null,"slug":"coin-counterfactual-image-generation-for-vqa","title":"COIN: Counterfactual Image Generation for VQA Interpretation","date":"2022-01-10","arxiv_id":"2201.03342","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-attention-ai-to-translate-low","title":"Interactive Attention AI to translate low light photos to captions for night scene understanding in women safety","date":"2022-01-04","arxiv_id":"2201.00969","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-general-purpose-vision-systems-an-end","title":"Towards General Purpose Vision Systems: An End-to-End Task-Agnostic Vision-Language Architecture","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/transform-retrieve-generate-natural-language","slug":"transform-retrieve-generate-natural-language","title":"Transform-Retrieve-Generate: Natural Language-Centric Outside-Knowledge Visual Question Answering","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"v-doc-visual-questions-answers-with-documents-1","title":"V-Doc: Visual Questions Answers With Documents","date":"2022-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"2112-13906","title":"Does CLIP Benefit Visual Question Answering in the Medical Domain as Much as it Does in the General Domain?","date":"2021-12-27","arxiv_id":"2112.13906","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-oriented-multi-user-semantic-1","title":"Task-Oriented Multi-User Semantic Communications","date":"2021-12-19","arxiv_id":"2112.10255","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resource-learning-with-knowledge-graphs-a","title":"Zero-shot and Few-shot Learning with Knowledge Graphs: A Comprehensive Survey","date":"2021-12-18","arxiv_id":"2112.10006","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-attention-for-vision-and","title":"Understanding Attention for Vision-and-Language Tasks","date":"2021-12-17","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-question-answering","title":"3D Question Answering","date":"2021-12-15","arxiv_id":"2112.08359","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-and-diagnosing-knowledge-based","title":"Improving and Diagnosing Knowledge-Based Visual Question Answering via Entity Enhanced Knowledge Injection","date":"2021-12-13","arxiv_id":"2112.06888","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-multimodal-pre-training-and-prompt","title":"Unified Multimodal Pre-training and Prompt-based Tuning for Vision-Language Understanding and Generation","date":"2021-12-10","arxiv_id":"2112.05587","repositories_listed":0,"syntology":null},{"url":null,"slug":"moca-incorporating-multi-stage-domain","title":"MoCA: Incorporating Multi-stage Domain Pretraining and Cross-guided Multimodal Attention for Textbook Question Answering","date":"2021-12-06","arxiv_id":"2112.02839","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-effectively-improves-low","title":"Curriculum Learning Effectively Improves Low Data VQA","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"80573d3f864dd36955d447f5eaff6cd8e2b7d32026792e1baa3fa8299e1f985c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}