{"url":"/method/deepvit","slug":"deepvit","name":"DeepViT","full_name":"DeepViT","full_name_withheld":false,"description_markdown":"**DeepViT** is a type of [vision transformer](https://paperswithcode.com/method/vision-transformer) that replaces the self-attention layer within the [transformer](https://paperswithcode.com/method/transformer) block with a [Re-attention module](https://paperswithcode.com/method/re-attention-module) to address the issue of attention collapse and enables training deeper ViTs.","description_state":"present","introduced_year":null,"introduced_by":{"title":"DeepViT: Towards Deeper Vision Transformer","paper":"/paper/deepvit-towards-deeper-vision-transformer","first_author":"Daquan Zhou","n_authors":8,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/deepvit-towards-deeper-vision-transformer"},"source":{"url":"https://arxiv.org/abs/2103.11886v4","title":"DeepViT: Towards Deeper Vision Transformer","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/zhoudaquan/dvit_repo/blob/f652676687384074a891e967c26e0a5daf5663a2/models/deep_vision_transformer.py#L55","code_snippet_url_on_a_code_host":true,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Image Models","url":"/methods/category/image-models","pwc_aliases":[]},{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision Transformers","url":"/methods/category/vision-transformers","pwc_aliases":["vision-transformer"]}],"n_papers_tagged":2,"archive_num_papers":2,"papers_newest_first":[{"paper":"/paper/on-the-calibration-of-human-pose-estimation","title":"On the Calibration of Human Pose Estimation","date":"2023-11-28","arxiv_id":"2311.17105","n_code_links":0,"syntology":null},{"paper":"/paper/deepvit-towards-deeper-vision-transformer","title":"DeepViT: Towards Deeper Vision Transformer","date":"2021-03-22","arxiv_id":"2103.11886","n_code_links":5,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}}],"papers_shown":2,"tasks":[{"task":"/task/2d-human-pose-estimation","name":"2D Human Pose Estimation","papers":1},{"task":"/task/human-mesh-recovery","name":"Human Mesh Recovery","papers":1},{"task":"/task/image-classification","name":"Image Classification","papers":1},{"task":"/task/pose-estimation","name":"Pose Estimation","papers":1},{"task":"/task/representation-learning","name":"Representation Learning","papers":1},{"task":"/task/image-classification","name":"image-classification","papers":1}],"tasks_shown":6,"n_tasks":6,"usage_by_year":[{"year":"2021","papers":1},{"year":"2023","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/deepvit"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}