{"url":"/method/vilt","slug":"vilt","name":"ViLT","full_name":"Vision-and-Language Transformer","full_name_withheld":false,"description_markdown":"ViLT is a minimal vision-and-language pre-training transformer model where processing of visual inputs is simplified to just the same convolution-free manner that text inputs are processed. The model-specific components of ViLT require less computation than the transformer component for multimodal interactions. ViLTThe model is pre-trained on the following objectives: image text matching, masked language modeling, and word patch alignment.","description_state":"present","introduced_year":null,"introduced_by":{"title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision","paper":"/paper/vilt-vision-and-language-transformer-without","first_author":"Wonjae Kim","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/vilt-vision-and-language-transformer-without"},"source":{"url":"https://arxiv.org/abs/2102.03334v2","title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision and Language Pre-Trained Models","url":"/methods/category/vision-and-language-pre-trained-models","pwc_aliases":[]}],"n_papers_tagged":6,"archive_num_papers":6,"papers_newest_first":[{"paper":null,"title":"Seeing More with Less: Human-like Representations in Vision Models","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/visual-robustness-benchmark-for-visual","title":"Visual Robustness Benchmark for Visual Question Answering (VQA)","date":"2024-07-03","arxiv_id":"2407.03386","n_code_links":1,"syntology":null},{"paper":"/paper/implicit-differentiable-outlier-detection","title":"Implicit Differentiable Outlier Detection Enable Robust Deep Multimodal Analysis","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/vl-checklist-evaluating-pre-trained-vision","title":"VL-CheckList: Evaluating Pre-trained Vision-Language Models with Objects, Attributes and Relations","date":"2022-07-01","arxiv_id":"2207.00221","n_code_links":1,"syntology":null},{"paper":"/paper/visual-spatial-reasoning","title":"Visual Spatial Reasoning","date":"2022-04-30","arxiv_id":"2205.00363","n_code_links":4,"syntology":{"ran":5,"of":10,"unverified":5,"pointer_only":0}},{"paper":"/paper/vilt-vision-and-language-transformer-without","title":"ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision","date":"2021-02-05","arxiv_id":"2102.03334","n_code_links":6,"syntology":{"ran":1,"of":4,"unverified":3,"pointer_only":1}}],"papers_shown":6,"tasks":[{"task":"/task/visual-question-answering","name":"Visual Question Answering (VQA)","papers":3},{"task":"/task/visual-reasoning","name":"Visual Reasoning","papers":3},{"task":"/task/cross-modal-retrieval","name":"Cross-Modal Retrieval","papers":2},{"task":"/task/image-retrieval","name":"Image Retrieval","papers":2},{"task":"/task/zero-shot-cross-modal-retrieval","name":"Zero-Shot Cross-Modal Retrieval","papers":2},{"task":"/task/object-detection-1","name":"object-detection","papers":2},{"task":"/task/image-captioning","name":"Image Captioning","papers":1},{"task":"/task/multimodal-intent-recognition","name":"Multimodal Intent Recognition","papers":1},{"task":"/task/natural-language-understanding","name":"Natural Language Understanding","papers":1},{"task":"/task/object-detection","name":"Object Detection","papers":1},{"task":"/task/question-answering","name":"Question Answering","papers":1},{"task":"/task/spatial-reasoning","name":"Spatial Reasoning","papers":1},{"task":"/task/text-retrieval","name":"Text Retrieval","papers":1},{"task":"/task/visual-entailment","name":"Visual Entailment","papers":1},{"task":"/task/visual-question-answering-1","name":"Visual Question Answering","papers":1},{"task":"/task/zero-shot-visual-question-answring","name":"Zero-Shot Visual Question Answring","papers":1},{"task":"/task/multimodal-interaction","name":"multimodal interaction","papers":1}],"tasks_shown":17,"n_tasks":17,"usage_by_year":[{"year":"2021","papers":1},{"year":"2022","papers":2},{"year":"2023","papers":1},{"year":"2024","papers":1},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/vilt"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}