{"url":"/method/esvit","slug":"esvit","name":"EsViT","full_name":"EsViT","full_name_withheld":false,"description_markdown":"**EsViT** proposes two techniques for developing efficient self-supervised vision transformers for visual representation leaning: a multi-stage architecture with sparse self-attention and a new pre-training task of region matching. The multi-stage architecture reduces modeling complexity but with a cost of losing the ability to capture fine-grained correspondences between image regions. The new pretraining task allows the model to capture fine-grained region dependencies and as a result significantly improves the quality of the learned vision representations.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Efficient Self-supervised Vision Transformers for Representation Learning","paper":"/paper/efficient-self-supervised-vision-transformers","first_author":"Chunyuan Li","n_authors":8,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/efficient-self-supervised-vision-transformers"},"source":{"url":"https://arxiv.org/abs/2106.09785v2","title":"Efficient Self-supervised Vision Transformers for Representation Learning","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision Transformers","url":"/methods/category/vision-transformers","pwc_aliases":["vision-transformer"]}],"n_papers_tagged":1,"archive_num_papers":1,"papers_newest_first":[{"paper":"/paper/efficient-self-supervised-vision-transformers","title":"Efficient Self-supervised Vision Transformers for Representation Learning","date":"2021-06-17","arxiv_id":"2106.09785","n_code_links":1,"syntology":{"ran":2,"of":6,"unverified":4,"pointer_only":0}}],"papers_shown":1,"tasks":[{"task":"/task/representation-learning","name":"Representation Learning","papers":1},{"task":"/task/self-supervised-image-classification","name":"Self-Supervised Image Classification","papers":1}],"tasks_shown":2,"n_tasks":2,"usage_by_year":[{"year":"2021","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/esvit"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}