{"url":"/method/t2t-vit","slug":"t2t-vit","name":"T2T-ViT","full_name":"Tokens-To-Token Vision Transformer","full_name_withheld":false,"description_markdown":"**T2T-ViT** (Tokens-To-Token Vision Transformer) is a type of [Vision Transformer](https://paperswithcode.com/method/vision-transformer) which incorporates 1) a layerwise Tokens-to-Token (T2T) transformation to progressively structurize the image to tokens by recursively aggregating neighboring Tokens into one Token (Tokens-to-Token), such that local structure represented by surrounding tokens can be modeled and tokens length can be reduced; 2) an efficient backbone with a deep-narrow structure for vision [transformer](https://paperswithcode.com/method/transformer) motivated by CNN architecture design after empirical study.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"https://arxiv.org/abs/2101.11986v3","title":"Tokens-to-Token ViT: Training Vision Transformers from Scratch on ImageNet","url_on_a_paper_host":true},"code_snippet_url":"https://github.com/yitu-opensource/T2T-ViT/blob/main/models/t2t_vit.py","code_snippet_url_on_a_code_host":true,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision Transformers","url":"/methods/category/vision-transformers","pwc_aliases":["vision-transformer"]}],"n_papers_tagged":11,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"Changing Base Without Losing Pace: A GPU-Efficient Alternative to MatMul in DNNs","date":"2025-03-15","arxiv_id":"2503.12211","n_code_links":0,"syntology":null},{"paper":"/paper/multi-criteria-token-fusion-with-one-step","title":"Multi-criteria Token Fusion with One-step-ahead Attention for Efficient Vision Transformers","date":"2024-03-15","arxiv_id":"2403.10030","n_code_links":1,"syntology":{"ran":3,"of":5,"unverified":2,"pointer_only":0}},{"paper":null,"title":"USDC: Unified Static and Dynamic Compression for Visual Transformer","date":"2023-10-17","arxiv_id":"2310.11117","n_code_links":0,"syntology":null},{"paper":"/paper/kdeformer-accelerating-transformers-via","title":"KDEformer: Accelerating Transformers via Kernel Density Estimation","date":"2023-02-05","arxiv_id":"2302.02451","n_code_links":1,"syntology":{"ran":8,"of":12,"unverified":4,"pointer_only":0}},{"paper":"/paper/unified-visual-transformer-compression-1","title":"Unified Visual Transformer Compression","date":"2022-03-15","arxiv_id":"2203.08243","n_code_links":1,"syntology":{"ran":9,"of":16,"unverified":7,"pointer_only":3}},{"paper":"/paper/bvit-broad-attention-based-vision-transformer","title":"BViT: Broad Attention based Vision Transformer","date":"2022-02-13","arxiv_id":"2202.06268","n_code_links":1,"syntology":null},{"paper":"/paper/multi-dimensional-model-compression-of-vision","title":"Multi-Dimensional Model Compression of Vision Transformer","date":"2021-12-31","arxiv_id":"2201.00043","n_code_links":1,"syntology":null},{"paper":"/paper/dynamic-token-normalization-improves-vision-1","title":"Dynamic Token Normalization Improves Vision Transformers","date":"2021-12-05","arxiv_id":"2112.02624","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":1}},{"paper":"/paper/scatterbrain-unifying-sparse-and-low-rank","title":"Scatterbrain: Unifying Sparse and Low-rank Attention Approximation","date":"2021-10-28","arxiv_id":"2110.15343","n_code_links":1,"syntology":{"ran":1,"of":1,"unverified":0,"pointer_only":0}},{"paper":"/paper/scatterbrain-unifying-sparse-and-low-rank-1","title":"Scatterbrain: Unifying Sparse and Low-rank Attention","date":"2021-05-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/tokens-to-token-vit-training-vision","title":"Tokens-to-Token ViT: Training Vision Transformers from Scratch on ImageNet","date":"2021-01-28","arxiv_id":"2101.11986","n_code_links":13,"syntology":{"ran":21,"of":26,"unverified":5,"pointer_only":8}}],"papers_shown":11,"tasks":[{"task":"/task/image-classification","name":"Image Classification","papers":3},{"task":"/task/image-generation","name":"Image Generation","papers":3},{"task":"/task/language-modeling","name":"Language Modeling","papers":3},{"task":"/task/language-modelling","name":"Language Modelling","papers":3},{"task":"/task/image-classification","name":"image-classification","papers":3},{"task":"/task/model-compression","name":"Model Compression","papers":2},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":1},{"task":"/task/density-estimation","name":"Density Estimation","papers":1},{"task":"/task/efficient-vits","name":"Efficient ViTs","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/inductive-bias","name":"Inductive Bias","papers":1},{"task":"/task/knowledge-distillation","name":"Knowledge Distillation","papers":1},{"task":"/task/listops","name":"ListOps","papers":1},{"task":"/task/object-detection","name":"Object Detection","papers":1},{"task":"/task/object-recognition","name":"Object Recognition","papers":1},{"task":"/task/model","name":"model","papers":1},{"task":"/task/object-detection-1","name":"object-detection","papers":1}],"tasks_shown":17,"n_tasks":17,"usage_by_year":[{"year":"2021","papers":5},{"year":"2022","papers":2},{"year":"2023","papers":2},{"year":"2024","papers":1},{"year":"2025","papers":1}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/t2t-vit"},"syntology_read_at":"2026-09-25T09:33:49+00:00"}