{"url":"/method/interbert","slug":"interbert","name":"InterBERT","full_name":"InterBERT","full_name_withheld":false,"description_markdown":"InterBERT aims to model interaction between information flows pertaining to different modalities. This new architecture builds multi-modal interaction and preserves the independence of single modal representation. InterBERT is built with an image embedding layer, a text embedding layer, a single-stream interaction module, and a two stream extraction module. The model is pre-trained with three tasks: 1) masked segment modeling, 2) masked region modeling, and 3) image-text matching.","description_state":"present","introduced_year":null,"introduced_by":{"title":"InterBERT: Vision-and-Language Interaction for Multi-modal Pretraining","paper":"/paper/interbert-vision-and-language-interaction-for","first_author":"Junyang Lin","n_authors":6,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/interbert-vision-and-language-interaction-for"},"source":{"url":"https://arxiv.org/abs/2003.13198v4","title":"InterBERT: Vision-and-Language Interaction for Multi-modal Pretraining","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision and Language Pre-Trained Models","url":"/methods/category/vision-and-language-pre-trained-models","pwc_aliases":[]}],"n_papers_tagged":1,"archive_num_papers":1,"papers_newest_first":[{"paper":"/paper/interbert-vision-and-language-interaction-for","title":"InterBERT: Vision-and-Language Interaction for Multi-modal Pretraining","date":"2020-03-30","arxiv_id":"2003.13198","n_code_links":0,"syntology":null}],"papers_shown":1,"tasks":[{"task":"/task/image-retrieval","name":"Image Retrieval","papers":1},{"task":"/task/image-text-matching","name":"Image-text matching","papers":1},{"task":"/task/retrieval","name":"Retrieval","papers":1},{"task":"/task/text-matching","name":"Text Matching","papers":1},{"task":"/task/visual-commonsense-reasoning","name":"Visual Commonsense Reasoning","papers":1}],"tasks_shown":5,"n_tasks":5,"usage_by_year":[{"year":"2020","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/interbert"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}