{"url":"/task/efficient-vits","name":"Efficient ViTs","slug":"efficient-vits","description_markdown":"Increasing the efficiency of ViTs without the modification of the architecture. (i.e., Key & Query Sparsification, Token pruning & merging)","categories":[{"name":"Adversarial","url":"/area/adversarial"},{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"derived"},"counts":{"papers_tagged":32,"papers_with_code":27,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":0,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/efficient-vits-on-imagenet-1k-with-deit-s","slug":"efficient-vits-on-imagenet-1k-with-deit-s","dataset":"ImageNet-1K (with DeiT-S)","dataset_url":null,"rows_in_archive":41,"metrics":["Top 1 Accuracy","GFLOPs"],"first_row_in_archive_order":{"model":"MCTF ($r=16$)","paper_title":"Multi-criteria Token Fusion with One-step-ahead Attention for Efficient Vision Transformers","paper_url":"/paper/multi-criteria-token-fusion-with-one-step","paper_date":"2024-03-15","arxiv_id":"2403.10030","code_links":[{"title":"mlvlab/mctf","url":"https://github.com/mlvlab/mctf"}],"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/efficient-vits-on-imagenet-1k-with-deit-t","slug":"efficient-vits-on-imagenet-1k-with-deit-t","dataset":"ImageNet-1K (with DeiT-T)","dataset_url":null,"rows_in_archive":22,"metrics":["Top 1 Accuracy","GFLOPs"],"first_row_in_archive_order":{"model":"dTPS","paper_title":"Joint Token Pruning and Squeezing Towards More Aggressive Compression of Vision Transformers","paper_url":"/paper/joint-token-pruning-and-squeezing-towards","paper_date":"2023-04-21","arxiv_id":"2304.10716","code_links":[{"title":"megvii-research/tps-cvpr2023","url":"https://github.com/megvii-research/tps-cvpr2023"}],"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":0}}},{"leaderboard":"/sota/efficient-vits-on-imagenet-1k-with-lv-vit-s","slug":"efficient-vits-on-imagenet-1k-with-lv-vit-s","dataset":"ImageNet-1K (With LV-ViT-S)","dataset_url":null,"rows_in_archive":19,"metrics":["Top 1 Accuracy","GFLOPs"],"first_row_in_archive_order":{"model":"MCTF ($r=8$)","paper_title":"Multi-criteria Token Fusion with One-step-ahead Attention for Efficient Vision Transformers","paper_url":"/paper/multi-criteria-token-fusion-with-one-step","paper_date":"2024-03-15","arxiv_id":"2403.10030","code_links":[{"title":"mlvlab/mctf","url":"https://github.com/mlvlab/mctf"}],"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":0}}}],"datasets":[],"subtasks":[],"parent_tasks":[{"url":"/task/image-classification","name":"Image Classification"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":27,"of":27,"tagged_in_all":32,"items":[{"url":"/paper/training-data-efficient-image-transformers","title":"Training data-efficient image transformers & distillation through attention","date":"2020-12-23","arxiv_id":"2012.12877","repositories_listed":40,"syntology":{"n":19,"n_ran":12,"n_unverified":7,"n_pointer_only":3}},{"url":"/paper/token-labeling-training-a-85-5-top-1-accuracy","title":"All Tokens Matter: Token Labeling for Training Better Vision Transformers","date":"2021-04-22","arxiv_id":"2104.10858","repositories_listed":7,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/token-merging-your-vit-but-faster","title":"Token Merging: Your ViT But Faster","date":"2022-10-17","arxiv_id":"2210.09461","repositories_listed":5,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":4}},{"url":"/paper/fast-vision-transformers-with-hilo-attention","title":"Fast Vision Transformers with HiLo Attention","date":"2022-05-26","arxiv_id":"2205.13213","repositories_listed":5,"syntology":{"n":13,"n_ran":2,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/castling-vit-compressing-self-attention-via","title":"Castling-ViT: Compressing Self-Attention via Switching Towards Linear-Angular Attention at Vision Transformer Inference","date":"2022-11-18","arxiv_id":"2211.10526","repositories_listed":4,"syntology":null},{"url":"/paper/pruning-self-attentions-into-convolutional","title":"Pruning Self-attentions into Convolutional Layers in Single Path","date":"2021-11-23","arxiv_id":"2111.11802","repositories_listed":3,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/ppt-token-pruning-and-pooling-for-efficient","title":"PPT: Token Pruning and Pooling for Efficient Vision Transformers","date":"2023-10-03","arxiv_id":"2310.01812","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/mdvit-multi-domain-vision-transformer-for","title":"MDViT: Multi-domain Vision Transformer for Small Medical Image Segmentation Datasets","date":"2023-07-05","arxiv_id":"2307.02100","repositories_listed":2,"syntology":null},{"url":"/paper/not-all-patches-are-what-you-need-expediting","title":"Not All Patches are What You Need: Expediting Vision Transformers via Token Reorganizations","date":"2022-02-16","arxiv_id":"2202.07800","repositories_listed":2,"syntology":{"n":15,"n_ran":6,"n_unverified":9,"n_pointer_only":12}},{"url":"/paper/dynamicvit-efficient-vision-transformers-with","title":"DynamicViT: Efficient Vision Transformers with Dynamic Token Sparsification","date":"2021-06-03","arxiv_id":"2106.02034","repositories_listed":2,"syntology":{"n":9,"n_ran":6,"n_unverified":3,"n_pointer_only":1}},{"url":"/paper/scalable-visual-transformers-with","title":"Scalable Vision Transformers with Hierarchical Pooling","date":"2021-03-19","arxiv_id":"2103.10619","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/trio-vit-post-training-quantization-and","title":"Trio-ViT: Post-Training Quantization and Acceleration for Softmax-Free Efficient Vision Transformer","date":"2024-05-06","arxiv_id":"2405.03882","repositories_listed":1,"syntology":null},{"url":"/paper/multi-criteria-token-fusion-with-one-step","title":"Multi-criteria Token Fusion with One-step-ahead Attention for Efficient Vision Transformers","date":"2024-03-15","arxiv_id":"2403.10030","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/gtp-vit-efficient-vision-transformers-via","title":"GTP-ViT: Efficient Vision Transformers via Graph-based Token Propagation","date":"2023-11-06","arxiv_id":"2311.03035","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/learned-thresholds-token-merging-and-pruning","title":"Learned Thresholds Token Merging and Pruning for Vision Transformers","date":"2023-07-20","arxiv_id":"2307.10780","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":5}},{"url":"/paper/shiftaddvit-mixture-of-multiplication-1","title":"ShiftAddViT: Mixture of Multiplication Primitives Towards Efficient Vision Transformer","date":"2023-06-10","arxiv_id":"2306.06446","repositories_listed":1,"syntology":{"n":20,"n_ran":8,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/diffrate-differentiable-compression-rate-for","title":"DiffRate : Differentiable Compression Rate for Efficient Vision Transformers","date":"2023-05-29","arxiv_id":"2305.17997","repositories_listed":1,"syntology":null},{"url":"/paper/joint-token-pruning-and-squeezing-towards","title":"Joint Token Pruning and Squeezing Towards More Aggressive Compression of Vision Transformers","date":"2023-04-21","arxiv_id":"2304.10716","repositories_listed":1,"syntology":{"n":8,"n_ran":2,"n_unverified":6,"n_pointer_only":0}},{"url":"/paper/making-vision-transformers-efficient-from-a","title":"Making Vision Transformers Efficient from A Token Sparsification View","date":"2023-03-15","arxiv_id":"2303.08685","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/beyond-attentive-tokens-incorporating-token","title":"Beyond Attentive Tokens: Incorporating Token Importance and Diversity for Efficient Vision Transformers","date":"2022-11-21","arxiv_id":"2211.11315","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-sparse-vit-towards-learnable","title":"Adaptive Sparse ViT: Towards Learnable Adaptive Token Pruning by Fully Exploiting Self-Attention","date":"2022-09-28","arxiv_id":"2209.13802","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/spvit-enabling-faster-vision-transformers-via","title":"SPViT: Enabling Faster Vision Transformers via Soft Token Pruning","date":"2021-12-27","arxiv_id":"2112.13890","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/adavit-adaptive-tokens-for-efficient-vision","title":"AdaViT: Adaptive Tokens for Efficient Vision Transformer","date":"2021-12-14","arxiv_id":"2112.07658","repositories_listed":1,"syntology":{"n":7,"n_ran":0,"n_unverified":7,"n_pointer_only":7}},{"url":"/paper/ats-adaptive-token-sampling-for-efficient","title":"Adaptive Token Sampling For Efficient Vision Transformers","date":"2021-11-30","arxiv_id":"2111.15667","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":5}},{"url":"/paper/nvit-vision-transformer-compression-and-1","title":"Global Vision Transformer Pruning with Hessian-Aware Saliency","date":"2021-10-10","arxiv_id":"2110.04869","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/evo-vit-slow-fast-token-evolution-for-dynamic","title":"Evo-ViT: Slow-Fast Token Evolution for Dynamic Vision Transformer","date":"2021-08-03","arxiv_id":"2108.01390","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/chasing-sparsity-in-vision-transformers-an","title":"Chasing Sparsity in Vision Transformers: An End-to-End Exploration","date":"2021-06-08","arxiv_id":"2106.04533","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_unverified":5,"n_pointer_only":0}}],"syntology_records":22,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}