{"url":"/method/unimo","slug":"unimo","name":"UNIMO","full_name":"UNIMO","full_name_withheld":false,"description_markdown":"**UNIMO** is a multi-modal pre-training architecture that can effectively adapt to both single modal and multimodal understanding and generation tasks. UNIMO learns visual representations and textual representations simultaneously, and unifies them into the same semantic space via [cross-modal contrastive learning](https://paperswithcode.com/method/cmcl) (CMCL) based on a large-scale corpus of image collections, text corpus and image-text pairs. The CMCL aligns the visual representation and textual representation, and unifies them into the same semantic\r\nspace based on image-text pairs.","description_state":"present","introduced_year":null,"introduced_by":{"title":"UNIMO: Towards Unified-Modal Understanding and Generation via Cross-Modal Contrastive Learning","paper":"/paper/unimo-towards-unified-modal-understanding-and","first_author":"Wei Li","n_authors":8,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/unimo-towards-unified-modal-understanding-and"},"source":{"url":"https://arxiv.org/abs/2012.15409v4","title":"UNIMO: Towards Unified-Modal Understanding and Generation via Cross-Modal Contrastive Learning","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision and Language Pre-Trained Models","url":"/methods/category/vision-and-language-pre-trained-models","pwc_aliases":[]},{"area":"Computer Vision","area_id":"computer-vision","collection":"Multi-Modal Methods","url":"/methods/category/multi-modal-methods","pwc_aliases":[]}],"n_papers_tagged":4,"archive_num_papers":4,"papers_newest_first":[{"paper":null,"title":"WuDaoMM: A large-scale Multi-Modal Dataset for Pre-training models","date":"2022-03-22","arxiv_id":"2203.11480","n_code_links":0,"syntology":null},{"paper":"/paper/unimo-2-end-to-end-unified-vision-language","title":"UNIMO-2: End-to-End Unified Vision-Language Grounded Learning","date":"2022-03-17","arxiv_id":"2203.09067","n_code_links":1,"syntology":null},{"paper":null,"title":"A Multimodal Sentiment Dataset for Video Recommendation","date":"2021-09-17","arxiv_id":"2109.08333","n_code_links":0,"syntology":null},{"paper":"/paper/unimo-towards-unified-modal-understanding-and","title":"UNIMO: Towards Unified-Modal Understanding and Generation via Cross-Modal Contrastive Learning","date":"2020-12-31","arxiv_id":"2012.15409","n_code_links":3,"syntology":null}],"papers_shown":4,"tasks":[{"task":"/task/image-captioning","name":"Image Captioning","papers":2},{"task":"/task/contrastive-learning","name":"Contrastive Learning","papers":1},{"task":"/task/cross-modal-retrieval","name":"Cross-Modal Retrieval","papers":1},{"task":"/task/image-generation","name":"Image Generation","papers":1},{"task":"/task/multimodal-sentiment-analysis","name":"Multimodal Sentiment Analysis","papers":1},{"task":"/task/question-answering","name":"Question Answering","papers":1},{"task":"/task/sentiment-analysis","name":"Sentiment Analysis","papers":1},{"task":"/task/text-to-image-generation-1","name":"Text to Image Generation","papers":1},{"task":"/task/text-to-image-generation","name":"Text-to-Image Generation","papers":1},{"task":"/task/video-understanding","name":"Video Understanding","papers":1},{"task":"/task/visual-question-answering","name":"Visual Question Answering (VQA)","papers":1}],"tasks_shown":11,"n_tasks":11,"usage_by_year":[{"year":"2020","papers":1},{"year":"2021","papers":1},{"year":"2022","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/unimo"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}