{"url":"/method/vl-bert","slug":"vl-bert","name":"VL-BERT","full_name":"Visual-Linguistic BERT","full_name_withheld":false,"description_markdown":"VL-BERT is pre-trained on a large-scale image-captions dataset together with text-only corpus. The input to the model are either words from the input sentences or regions-of-interest (RoI) from input images. It can be fine-tuned to fit most visual-linguistic downstream tasks. Its backbone is a multi-layer bidirectional Transformer encoder, modified to accommodate visual contents, and new type of visual feature embedding to the input feature embeddings. VL-BERT takes both visual and linguistic elements as input, represented as RoIs in images and subwords in input sentences. Four different types of embeddings are used to represent each input: token embedding, visual feature embedding, segment embedding, and sequence position embedding. VL-BERT is pre-trained using Conceptual Captions and text-only datasets. Two pre-training tasks are used: masked language modeling with visual clues, and masked RoI classification with linguistic clues.","description_state":"present","introduced_year":null,"introduced_by":{"title":"VL-BERT: Pre-training of Generic Visual-Linguistic Representations","paper":"/paper/vl-bert-pre-training-of-generic-visual","first_author":"Weijie Su","n_authors":7,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/vl-bert-pre-training-of-generic-visual"},"source":{"url":"https://arxiv.org/abs/1908.08530v4","title":"VL-BERT: Pre-training of Generic Visual-Linguistic Representations","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision and Language Pre-Trained Models","url":"/methods/category/vision-and-language-pre-trained-models","pwc_aliases":[]}],"n_papers_tagged":4,"archive_num_papers":4,"papers_newest_first":[{"paper":null,"title":"Self-Training Vision Language BERTs with a Unified Conditional Model","date":"2022-01-06","arxiv_id":"2201.02010","n_code_links":0,"syntology":null},{"paper":"/paper/bertgen-multi-task-generation-through-bert","title":"BERTGEN: Multi-task Generation through BERT","date":"2021-06-07","arxiv_id":"2106.03484","n_code_links":1,"syntology":null},{"paper":null,"title":"Worst of Both Worlds: Biases Compound in Pre-trained Vision-and-Language Models","date":"2021-04-18","arxiv_id":"2104.08666","n_code_links":0,"syntology":null},{"paper":"/paper/vl-bert-pre-training-of-generic-visual","title":"VL-BERT: Pre-training of Generic Visual-Linguistic Representations","date":"2019-08-22","arxiv_id":"1908.08530","n_code_links":3,"syntology":{"ran":0,"of":1,"unverified":1,"pointer_only":0}}],"papers_shown":4,"tasks":[{"task":"/task/decoder","name":"Decoder","papers":1},{"task":"/task/image-captioning","name":"Image Captioning","papers":1},{"task":"/task/image-text-matching","name":"Image-text matching","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/machine-translation","name":"Machine Translation","papers":1},{"task":"/task/multimodal-machine-translation","name":"Multimodal Machine Translation","papers":1},{"task":"/task/question-answering","name":"Question Answering","papers":1},{"task":"/task/referring-expression","name":"Referring Expression","papers":1},{"task":"/task/referring-expression-comprehension","name":"Referring Expression Comprehension","papers":1},{"task":"/task/sentence","name":"Sentence","papers":1},{"task":"/task/text-generation","name":"Text Generation","papers":1},{"task":"/task/translation","name":"Translation","papers":1},{"task":"/task/visual-commonsense-reasoning","name":"Visual Commonsense Reasoning","papers":1},{"task":"/task/visual-question-answering-1","name":"Visual Question Answering","papers":1},{"task":"/task/visual-question-answering","name":"Visual Question Answering (VQA)","papers":1}],"tasks_shown":15,"n_tasks":15,"usage_by_year":[{"year":"2019","papers":1},{"year":"2021","papers":2},{"year":"2022","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/vl-bert"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}