{"url":"/method/visualbert","slug":"visualbert","name":"VisualBERT","full_name":"VisualBERT","full_name_withheld":false,"description_markdown":"VisualBERT aims to reuse self-attention to implicitly align elements of the input text and regions in the input image. Visual embeddings are used to model images where the representations are represented by a bounding region in an image obtained from an object detector. These visual embeddings are constructed by summing three embeddings: 1) visual feature representation, 2) a segment embedding indicate whether it is an image embedding, and 3) position embedding. Essentially, image regions and language are combined with a Transformer to allow self-attention to discover implicit alignments between language and vision. VisualBERT is trained using COCO, which consists of images paired with captions. It is pre-trained using two objectives: masked language modeling objective and sentence-image prediction task. It can then be fine-tuned on different downstream tasks.","description_state":"present","introduced_year":null,"introduced_by":{"title":"VisualBERT: A Simple and Performant Baseline for Vision and Language","paper":"/paper/visualbert-a-simple-and-performant-baseline","first_author":"Liunian Harold Li","n_authors":5,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/visualbert-a-simple-and-performant-baseline"},"source":{"url":"https://arxiv.org/abs/1908.03557v1","title":"VisualBERT: A Simple and Performant Baseline for Vision and Language","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision and Language Pre-Trained Models","url":"/methods/category/vision-and-language-pre-trained-models","pwc_aliases":[]}],"n_papers_tagged":25,"archive_num_papers":25,"papers_newest_first":[{"paper":null,"title":"Visual Question Answering on Multiple Remote Sensing Image Modalities","date":"2025-05-21","arxiv_id":"2505.15401","n_code_links":0,"syntology":null},{"paper":null,"title":"Seeing Through VisualBERT: A Causal Adventure on Memetic Landscapes","date":"2024-10-17","arxiv_id":"2410.13488","n_code_links":0,"syntology":null},{"paper":null,"title":"OSPC: Detecting Harmful Memes with Large Language Model as a Catalyst","date":"2024-06-14","arxiv_id":"2406.09779","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-image-text-matching-verb-understanding","title":"Beyond Image-Text Matching: Verb Understanding in Multimodal Transformers Using Guided Masking","date":"2024-01-29","arxiv_id":"2401.16575","n_code_links":1,"syntology":null},{"paper":"/paper/a-review-of-vision-language-models-and-their","title":"A Review of Vision-Language Models and their Performance on the Hateful Memes Challenge","date":"2023-05-09","arxiv_id":"2305.06159","n_code_links":1,"syntology":null},{"paper":null,"title":"Controlling for Stereotypes in Multimodal Language Model Evaluation","date":"2023-02-03","arxiv_id":"2302.01582","n_code_links":0,"syntology":null},{"paper":null,"title":"A survey on knowledge-enhanced multimodal learning","date":"2022-11-19","arxiv_id":"2211.12328","n_code_links":0,"syntology":null},{"paper":"/paper/transfer-learning-with-joint-fine-tuning-for","title":"Transfer Learning with Joint Fine-Tuning for Multimodal Sentiment Analysis","date":"2022-10-11","arxiv_id":"2210.05790","n_code_links":1,"syntology":null},{"paper":null,"title":"Multi-Modal Fusion Transformer for Visual Question Answering in Remote Sensing","date":"2022-10-10","arxiv_id":"2210.04510","n_code_links":0,"syntology":null},{"paper":"/paper/surgical-vqa-visual-question-answering-in","title":"Surgical-VQA: Visual Question Answering in Surgical Scenes using Transformer","date":"2022-06-22","arxiv_id":"2206.11053","n_code_links":3,"syntology":{"ran":1,"of":2,"unverified":1,"pointer_only":2}},{"paper":"/paper/visual-spatial-reasoning","title":"Visual Spatial Reasoning","date":"2022-04-30","arxiv_id":"2205.00363","n_code_links":4,"syntology":{"ran":5,"of":10,"unverified":5,"pointer_only":0}},{"paper":null,"title":"Visio-Linguistic Brain Encoding","date":"2022-04-18","arxiv_id":"2204.08261","n_code_links":0,"syntology":null},{"paper":"/paper/hateful-memes-challenge-an-enhanced","title":"Hateful Memes Challenge: An Enhanced Multimodal Framework","date":"2021-12-20","arxiv_id":"2112.11244","n_code_links":1,"syntology":null},{"paper":null,"title":"Multimodal Learning: Are Captions All You Need?","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Seeing things or seeing scenes: Investigating the capabilities of V&L models to align scene descriptions to images","date":"2021-10-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Understanding of Emotion Perception from Art","date":"2021-10-13","arxiv_id":"2110.06486","n_code_links":0,"syntology":null},{"paper":"/paper/marmot-a-deep-learning-framework-for","title":"MARMOT: A Deep Learning Framework for Constructing Multimodal Representations for Vision-and-Language Tasks","date":"2021-09-23","arxiv_id":"2109.11526","n_code_links":1,"syntology":null},{"paper":null,"title":"What Vision-Language Models `See' when they See Scenes","date":"2021-09-15","arxiv_id":"2109.07301","n_code_links":0,"syntology":null},{"paper":"/paper/broaden-the-vision-geo-diverse-visual","title":"Broaden the Vision: Geo-Diverse Visual Commonsense Reasoning","date":"2021-09-14","arxiv_id":"2109.06860","n_code_links":1,"syntology":{"ran":0,"of":9,"unverified":9,"pointer_only":0}},{"paper":"/paper/berthop-an-effective-vision-and-language","title":"BERTHop: An Effective Vision-and-Language Model for Chest X-ray Disease Diagnosis","date":"2021-08-10","arxiv_id":"2108.04938","n_code_links":1,"syntology":null},{"paper":null,"title":"Chop Chop BERT: Visual Question Answering by Chopping VisualBERT's Heads","date":"2021-04-30","arxiv_id":"2104.14741","n_code_links":0,"syntology":null},{"paper":null,"title":"Analysis on Image Set Visual Question Answering","date":"2021-03-31","arxiv_id":"2104.00107","n_code_links":0,"syntology":null},{"paper":"/paper/detecting-hate-speech-in-memes-using","title":"Detecting Hate Speech in Memes Using Multimodal Deep Learning Approaches: Prize-winning solution to Hateful Memes Challenge","date":"2020-12-23","arxiv_id":"2012.12975","n_code_links":1,"syntology":null},{"paper":"/paper/a-comparison-of-pre-trained-vision-and","title":"A Comparison of Pre-trained Vision-and-Language Models for Multimodal Representation Learning across Medical Images and Reports","date":"2020-09-03","arxiv_id":"2009.01523","n_code_links":1,"syntology":null},{"paper":"/paper/visualbert-a-simple-and-performant-baseline","title":"VisualBERT: A Simple and Performant Baseline for Vision and Language","date":"2019-08-09","arxiv_id":"1908.03557","n_code_links":10,"syntology":{"ran":4,"of":9,"unverified":5,"pointer_only":6}}],"papers_shown":25,"tasks":[{"task":"/task/visual-question-answering","name":"Visual Question Answering (VQA)","papers":9},{"task":"/task/question-answering","name":"Question Answering","papers":7},{"task":"/task/visual-question-answering-1","name":"Visual Question Answering","papers":7},{"task":"/task/language-modeling","name":"Language Modeling","papers":4},{"task":"/task/language-modelling","name":"Language Modelling","papers":4},{"task":"/task/visual-reasoning","name":"Visual Reasoning","papers":4},{"task":"/task/image-captioning","name":"Image Captioning","papers":3},{"task":"/task/representation-learning","name":"Representation Learning","papers":3},{"task":"/task/multimodal-deep-learning","name":"Multimodal Deep Learning","papers":2},{"task":"/task/object","name":"Object","papers":2},{"task":"/task/visual-commonsense-reasoning","name":"Visual Commonsense Reasoning","papers":2},{"task":"/task/visual-entailment","name":"Visual Entailment","papers":2},{"task":"/task/all","name":"All","papers":1},{"task":"/task/conditional-image-generation","name":"Conditional Image Generation","papers":1},{"task":"/task/culture","name":"Cultural Vocal Bursts Intensity Prediction","papers":1},{"task":"/task/ensemble-learning","name":"Ensemble Learning","papers":1},{"task":"/task/factual-visual-question-answering","name":"Factual Visual Question Answering","papers":1},{"task":"/task/fairness","name":"Fairness","papers":1},{"task":"/task/image-text-retrieval","name":"Image-text Retrieval","papers":1},{"task":"/task/image-text-matching","name":"Image-text matching","papers":1}],"tasks_shown":20,"n_tasks":45,"usage_by_year":[{"year":"2019","papers":1},{"year":"2020","papers":2},{"year":"2021","papers":10},{"year":"2022","papers":6},{"year":"2023","papers":2},{"year":"2024","papers":3},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/visualbert"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}