{"url":"/method/coat","slug":"coat","name":"CoaT","full_name":"Co-Scale Conv-attentional Image Transformer","full_name_withheld":false,"description_markdown":"**Co-Scale Conv-Attentional Image Transformer** (CoaT) is a [Transformer](https://paperswithcode.com/method/transformer)-based image classifier equipped with co-scale and conv-attentional mechanisms. First, the co-scale mechanism maintains the integrity of Transformers' encoder branches at individual scales, while allowing representations learned at different scales to effectively communicate with each other. Second, the conv-attentional mechanism is designed by realizing a relative position embedding formulation in the factorized attention module with an efficient [convolution](https://paperswithcode.com/method/convolution)-like implementation. CoaT empowers image Transformers with enriched multi-scale and contextual modeling capabilities.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Co-Scale Conv-Attentional Image Transformers","paper":"/paper/co-scale-conv-attentional-image-transformers","first_author":"Weijian Xu","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/co-scale-conv-attentional-image-transformers"},"source":{"url":"https://arxiv.org/abs/2104.06399v2","title":"Co-Scale Conv-Attentional Image Transformers","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision Transformers","url":"/methods/category/vision-transformers","pwc_aliases":["vision-transformer"]}],"n_papers_tagged":5,"archive_num_papers":5,"papers_newest_first":[{"paper":null,"title":"Enhance Mobile Agents Thinking Process Via Iterative Preference Learning","date":"2025-05-18","arxiv_id":"2505.12299","n_code_links":0,"syntology":null},{"paper":"/paper/what-do-vision-transformers-learn-a-visual","title":"What do Vision Transformers Learn? A Visual Exploration","date":"2022-12-13","arxiv_id":"2212.06727","n_code_links":1,"syntology":null},{"paper":null,"title":"Exploring Advances in Transformers and CNN for Skin Lesion Diagnosis on Small Datasets","date":"2022-05-30","arxiv_id":"2205.15442","n_code_links":0,"syntology":null},{"paper":null,"title":"Exploring and Improving Mobile Level Vision Transformers","date":"2021-08-30","arxiv_id":"2108.13015","n_code_links":0,"syntology":null},{"paper":"/paper/co-scale-conv-attentional-image-transformers","title":"Co-Scale Conv-Attentional Image Transformers","date":"2021-04-13","arxiv_id":"2104.06399","n_code_links":9,"syntology":{"ran":17,"of":22,"unverified":5,"pointer_only":6}}],"papers_shown":5,"tasks":[{"task":"/task/continual-pretraining","name":"Continual Pretraining","papers":1},{"task":"/task/instance-segmentation","name":"Instance Segmentation","papers":1},{"task":"/task/language-modelling","name":"Language Modelling","papers":1},{"task":"/task/object-detection","name":"Object Detection","papers":1},{"task":"/task/semantic-segmentation","name":"Semantic Segmentation","papers":1},{"task":"/task/object-detection-1","name":"object-detection","papers":1}],"tasks_shown":6,"n_tasks":6,"usage_by_year":[{"year":"2021","papers":2},{"year":"2022","papers":2},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/coat"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}