{"url":"/method/flava","slug":"flava","name":"FLAVA","full_name":"FLAVA","full_name_withheld":false,"description_markdown":"FLAVA aims at building a single holistic universal model that targets all modalities at once. FLAVA is a language vision alignment model that learns strong representations from multimodal data (image-text pairs) and unimodal data (unpaired images and text). The model consists of an image encode transformer to capture unimodal image representations, a text encoder transformer to process unimodal text information, and a multimodal encode transformer that takes as input the encoded unimodal image and text and integrates their representations for multimodal reasoning. During pretraining, masked image modeling (MIM) and mask language modeling (MLM) losses are applied onto the image and text encoders over a single image or a text piece, respectively, while contrastive, masked multimodal modeling (MMM), and image-text matching (ITM) loss are used over paired image-text data. For downstream tasks, classification heads are applied on the outputs from the image, text, and multimodal encoders respectively for visual recognition, language understanding, and multimodal reasoning tasks It can be applied to broad scope of tasks from three domains (visual recognition, language understanding, and multimodal reasoning) under a common transformer model architecture.","description_state":"present","introduced_year":null,"introduced_by":{"title":"FLAVA: A Foundational Language And Vision Alignment Model","paper":"/paper/flava-a-foundational-language-and-vision","first_author":"Amanpreet Singh","n_authors":7,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/flava-a-foundational-language-and-vision"},"source":{"url":"https://arxiv.org/abs/2112.04482v3","title":"FLAVA: A Foundational Language And Vision Alignment Model","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision and Language Pre-Trained Models","url":"/methods/category/vision-and-language-pre-trained-models","pwc_aliases":[]}],"n_papers_tagged":11,"archive_num_papers":11,"papers_newest_first":[{"paper":null,"title":"The age of spiritual machines: Language quietus induces synthetic altered states of consciousness in artificial intelligence","date":"2024-09-30","arxiv_id":"2410.00257","n_code_links":0,"syntology":null},{"paper":null,"title":"OSPC: Detecting Harmful Memes with Large Language Model as a Catalyst","date":"2024-06-14","arxiv_id":"2406.09779","n_code_links":0,"syntology":null},{"paper":null,"title":"Interpreting the structure of multi-object representations in vision encoders","date":"2024-06-13","arxiv_id":"2406.09067","n_code_links":0,"syntology":null},{"paper":null,"title":"Acquiring Linguistic Knowledge from Multimodal Input","date":"2024-02-27","arxiv_id":"2402.17936","n_code_links":0,"syntology":null},{"paper":null,"title":"AiGen-FoodReview: A Multimodal Dataset of Machine-Generated Restaurant Reviews and Images on Social Media","date":"2024-01-16","arxiv_id":"2401.08825","n_code_links":0,"syntology":null},{"paper":"/paper/whisbert-multimodal-text-audio-language","title":"WhisBERT: Multimodal Text-Audio Language Modeling on 100M Words","date":"2023-12-05","arxiv_id":"2312.02931","n_code_links":1,"syntology":null},{"paper":null,"title":"STELLA: Continual Audio-Video Pre-training with Spatio-Temporal Localized Alignment","date":"2023-10-12","arxiv_id":"2310.08204","n_code_links":0,"syntology":null},{"paper":null,"title":"MoMo: A shared encoder Model for text, image and multi-Modal representations","date":"2023-04-11","arxiv_id":"2304.05523","n_code_links":0,"syntology":null},{"paper":null,"title":"Controlling for Stereotypes in Multimodal Language Model Evaluation","date":"2023-02-03","arxiv_id":"2302.01582","n_code_links":0,"syntology":null},{"paper":"/paper/vl-taboo-an-analysis-of-attribute-based-zero","title":"VL-Taboo: An Analysis of Attribute-based Zero-shot Capabilities of Vision-Language Models","date":"2022-09-12","arxiv_id":"2209.06103","n_code_links":1,"syntology":null},{"paper":"/paper/flava-a-foundational-language-and-vision","title":"FLAVA: A Foundational Language And Vision Alignment Model","date":"2021-12-08","arxiv_id":"2112.04482","n_code_links":4,"syntology":null}],"papers_shown":11,"tasks":[{"task":"/task/language-modelling","name":"Language Modelling","papers":4},{"task":"/task/language-modeling","name":"Language Modeling","papers":3},{"task":"/task/attribute","name":"Attribute","papers":1},{"task":"/task/continual-learning","name":"Continual Learning","papers":1},{"task":"/task/decision-making","name":"Decision Making","papers":1},{"task":"/task/image-captioning","name":"Image Captioning","papers":1},{"task":"/task/image-retrieval","name":"Image Retrieval","papers":1},{"task":"/task/image-text-retrieval","name":"Image-text Retrieval","papers":1},{"task":"/task/image-to-text-retrieval","name":"Image-to-Text Retrieval","papers":1},{"task":"/task/language-model-evaluation","name":"Language Model Evaluation","papers":1},{"task":"/task/large-language-model","name":"Large Language Model","papers":1},{"task":"/task/object","name":"Object","papers":1},{"task":"/task/optical-character-recognition","name":"Optical Character Recognition","papers":1},{"task":"/task/optical-character-recognition","name":"Optical Character Recognition (OCR)","papers":1},{"task":"/task/representation-learning","name":"Representation Learning","papers":1},{"task":"/task/retrieval","name":"Retrieval","papers":1},{"task":"/task/text-retrieval","name":"Text Retrieval","papers":1},{"task":"/task/unity","name":"Unity","papers":1},{"task":"/task/video-alignment","name":"Video Alignment","papers":1},{"task":"/task/visual-reasoning","name":"Visual Reasoning","papers":1}],"tasks_shown":20,"n_tasks":25,"usage_by_year":[{"year":"2021","papers":1},{"year":"2022","papers":1},{"year":"2023","papers":4},{"year":"2024","papers":5}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/flava"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}