{"url":"/method/altclip","slug":"altclip","name":"AltCLIP","full_name":"AltCLIP","full_name_withheld":false,"description_markdown":"In this work, we present a conceptually simple and effective method to train a strong bilingual multimodal representation model. Starting from the pretrained multimodal representation model CLIP released by OpenAI, we switched its text encoder with a pretrained multilingual text encoder XLM-R, and aligned both languages and image representations by a two-stage training schema consisting of teacher learning and contrastive learning. We validate our method through evaluations of a wide range of tasks. We set new state-of-the-art performances on a bunch of tasks including ImageNet-CN, Flicker30k- CN, and COCO-CN. Further, we obtain very close performances with CLIP on almost all tasks, suggesting that one can simply alter the text encoder in CLIP for extended capabilities such as multilingual understanding. Our models and code are available at https://github.com/FlagAI-Open/FlagAI.","description_state":"present","introduced_year":null,"introduced_by":{"title":"AltCLIP: Altering the Language Encoder in CLIP for Extended Language Capabilities","paper":"/paper/altclip-altering-the-language-encoder-in-clip","first_author":"Zhongzhi Chen","n_authors":6,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/altclip-altering-the-language-encoder-in-clip"},"source":{"url":"https://arxiv.org/abs/2211.06679v2","title":"AltCLIP: Altering the Language Encoder in CLIP for Extended Language Capabilities","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision and Language Pre-Trained Models","url":"/methods/category/vision-and-language-pre-trained-models","pwc_aliases":[]}],"n_papers_tagged":5,"archive_num_papers":5,"papers_newest_first":[{"paper":"/paper/sparse-vs-contiguous-adversarial-pixel","title":"Sparse vs Contiguous Adversarial Pixel Perturbations in Multimodal Models: An Empirical Analysis","date":"2024-07-25","arxiv_id":"2407.18251","n_code_links":1,"syntology":null},{"paper":null,"title":"Assessing Brittleness of Image-Text Retrieval Benchmarks from Vision-Language Models Perspective","date":"2024-07-21","arxiv_id":"2407.15239","n_code_links":0,"syntology":null},{"paper":null,"title":"A Progressive Framework of Vision-language Knowledge Distillation and Alignment for Multilingual Scene","date":"2024-04-17","arxiv_id":"2404.11249","n_code_links":0,"syntology":null},{"paper":"/paper/altdiffusion-a-multilingual-text-to-image","title":"AltDiffusion: A Multilingual Text-to-Image Diffusion Model","date":"2023-08-19","arxiv_id":"2308.09991","n_code_links":1,"syntology":{"ran":9,"of":11,"unverified":2,"pointer_only":11}},{"paper":"/paper/altclip-altering-the-language-encoder-in-clip","title":"AltCLIP: Altering the Language Encoder in CLIP for Extended Language Capabilities","date":"2022-11-12","arxiv_id":"2211.06679","n_code_links":2,"syntology":{"ran":2,"of":11,"unverified":9,"pointer_only":0}}],"papers_shown":5,"tasks":[{"task":"/task/image-classification","name":"Image Classification","papers":2},{"task":"/task/knowledge-distillation","name":"Knowledge Distillation","papers":2},{"task":"/task/zero-shot-image-classification","name":"Zero-Shot Image Classification","papers":2},{"task":"/task/blocking","name":"Blocking","papers":1},{"task":"/task/concept-alignment","name":"Concept Alignment","papers":1},{"task":"/task/contrastive-learning","name":"Contrastive Learning","papers":1},{"task":"/task/cross-modal-retrieval","name":"Cross-Modal Retrieval","papers":1},{"task":"/task/image-retrieval","name":"Image Retrieval","papers":1},{"task":"/task/image-text-retrieval","name":"Image-text Retrieval","papers":1},{"task":"/task/image-to-text-retrieval","name":"Image-to-Text Retrieval","papers":1},{"task":"/task/information-retrieval","name":"Information Retrieval","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/retrieval","name":"Retrieval","papers":1},{"task":"/task/text-retrieval","name":"Text Retrieval","papers":1},{"task":"/task/text-to-image-generation","name":"Text-to-Image Generation","papers":1},{"task":"/task/xlm-r","name":"XLM-R","papers":1},{"task":"/task/zero-shot-cross-modal-retrieval","name":"Zero-Shot Cross-Modal Retrieval","papers":1},{"task":"/task/zero-shot-transfer-image-classification","name":"Zero-Shot Transfer Image Classification","papers":1},{"task":"/task/zero-shot-transfer-image-classification-cn","name":"Zero-Shot Transfer Image Classification (CN)","papers":1},{"task":"/task/zero-shot-image-retrieval","name":"Zero-shot Image Retrieval","papers":1}],"tasks_shown":20,"n_tasks":23,"usage_by_year":[{"year":"2022","papers":1},{"year":"2023","papers":1},{"year":"2024","papers":3}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/altclip"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}