{"url":"/method/simvlm","slug":"simvlm","name":"SimVLM","full_name":"Simple Visual Language Model","full_name_withheld":false,"description_markdown":"SimVLM is a minimalist pretraining framework to reduce training complexity by exploiting large-scale weak supervision. It is trained end-to-end with a single prefix language modeling (PrefixLM) objective. PrefixLM enables bidirectional attention within the prefix sequence, and thus it is applicable for both decoder-only\r\nand encoder-decoder sequence-to-sequence language models.","description_state":"present","introduced_year":null,"introduced_by":{"title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision","paper":"/paper/simvlm-simple-visual-language-model","first_author":"ZiRui Wang","n_authors":6,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/simvlm-simple-visual-language-model"},"source":{"url":"https://arxiv.org/abs/2108.10904v3","title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision and Language Pre-Trained Models","url":"/methods/category/vision-and-language-pre-trained-models","pwc_aliases":[]}],"n_papers_tagged":3,"archive_num_papers":3,"papers_newest_first":[{"paper":"/paper/coca-contrastive-captioners-are-image-text","title":"CoCa: Contrastive Captioners are Image-Text Foundation Models","date":"2022-05-04","arxiv_id":"2205.01917","n_code_links":6,"syntology":{"ran":9,"of":17,"unverified":8,"pointer_only":0}},{"paper":"/paper/magma-multimodal-augmentation-of-generative","title":"MAGMA -- Multimodal Augmentation of Generative Models through Adapter-based Finetuning","date":"2021-12-09","arxiv_id":"2112.05253","n_code_links":1,"syntology":{"ran":10,"of":15,"unverified":5,"pointer_only":0}},{"paper":"/paper/simvlm-simple-visual-language-model","title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision","date":"2021-08-24","arxiv_id":"2108.10904","n_code_links":2,"syntology":{"ran":18,"of":37,"unverified":19,"pointer_only":28}}],"papers_shown":3,"tasks":[{"task":"/task/image-captioning","name":"Image Captioning","papers":2},{"task":"/task/language-modeling","name":"Language Modeling","papers":2},{"task":"/task/language-modelling","name":"Language Modelling","papers":2},{"task":"/task/visual-question-answering-1","name":"Visual Question Answering","papers":2},{"task":"/task/visual-question-answering","name":"Visual Question Answering (VQA)","papers":2},{"task":"/task/action-classification","name":"Action Classification","papers":1},{"task":"/task/decoder","name":"Decoder","papers":1},{"task":"/task/image-classification","name":"Image Classification","papers":1},{"task":"/task/in-context-learning","name":"In-Context Learning","papers":1},{"task":"/task/question-answering","name":"Question Answering","papers":1},{"task":"/task/representation-learning","name":"Representation Learning","papers":1},{"task":"/task/retrieval","name":"Retrieval","papers":1},{"task":"/task/video-retrieval","name":"Video Retrieval","papers":1},{"task":"/task/visual-entailment","name":"Visual Entailment","papers":1},{"task":"/task/visual-reasoning","name":"Visual Reasoning","papers":1},{"task":"/task/zero-shot-cross-modal-retrieval","name":"Zero-Shot Cross-Modal Retrieval","papers":1},{"task":"/task/zero-shot-transfer-image-classification","name":"Zero-Shot Transfer Image Classification","papers":1}],"tasks_shown":17,"n_tasks":17,"usage_by_year":[{"year":"2021","papers":2},{"year":"2022","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/simvlm"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}