{"url":"/method/bottleneck-transformer","slug":"bottleneck-transformer","name":"Bottleneck Transformer","full_name":"Bottleneck Transformer","full_name_withheld":false,"description_markdown":"The **Bottleneck Transformer (BoTNet) ** is an image classification model that incorporates self-attention for multiple computer vision tasks including image classification, object detection and instance segmentation. By just replacing the spatial convolutions with global self-attention in the final three bottleneck blocks of a [ResNet](https://paperswithcode.com/method/resnet) and no other changes, the approach improves upon baselines significantly on instance segmentation and object detection while also reducing the parameters, with minimal overhead in latency.","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":"https://arxiv.org/abs/2101.11605v2","title":"Bottleneck Transformers for Visual Recognition","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Image Models","url":"/methods/category/image-models","pwc_aliases":[]}],"n_papers_tagged":10,"archive_num_papers":null,"papers_newest_first":[{"paper":"/paper/robust-multimodal-survival-prediction-with","title":"Robust Multimodal Survival Prediction with the Latent Differentiation Conditional Variational AutoEncoder","date":"2025-03-12","arxiv_id":"2503.09496","n_code_links":1,"syntology":{"ran":6,"of":6,"unverified":0,"pointer_only":6}},{"paper":null,"title":"Robust Multimodal Survival Prediction with Conditional Latent Differentiation Variational AutoEncoder","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/multi-scale-bottleneck-transformer-for-weakly","title":"Multi-scale Bottleneck Transformer for Weakly Supervised Multimodal Violence Detection","date":"2024-05-08","arxiv_id":"2405.05130","n_code_links":1,"syntology":null},{"paper":null,"title":"Rock Classification Based on Residual Networks","date":"2024-02-19","arxiv_id":"2402.11831","n_code_links":0,"syntology":null},{"paper":"/paper/svfap-self-supervised-video-facial-affect","title":"SVFAP: Self-supervised Video Facial Affect Perceiver","date":"2023-12-31","arxiv_id":"2401.00416","n_code_links":1,"syntology":null},{"paper":"/paper/learning-bottleneck-transformer-for-event","title":"Learning Bottleneck Transformer for Event Image-Voxel Feature Fusion based Classification","date":"2023-08-23","arxiv_id":"2308.11937","n_code_links":1,"syntology":{"ran":5,"of":6,"unverified":1,"pointer_only":6}},{"paper":null,"title":"Marine Debris Detection in Satellite Surveillance using Attention Mechanisms","date":"2023-07-09","arxiv_id":"2307.04128","n_code_links":0,"syntology":null},{"paper":null,"title":"Cross-Domain Synthetic-to-Real In-the-Wild Depth and Normal Estimation for 3D Scene Understanding","date":"2022-12-09","arxiv_id":"2212.05040","n_code_links":0,"syntology":null},{"paper":"/paper/anatomy-guided-parallel-bottleneck","title":"AGMB-Transformer: Anatomy-Guided Multi-Branch Transformer Network for Automated Evaluation of Root Canal Therapy","date":"2021-05-02","arxiv_id":"2105.00381","n_code_links":1,"syntology":null},{"paper":"/paper/bottleneck-transformers-for-visual","title":"Bottleneck Transformers for Visual Recognition","date":"2021-01-27","arxiv_id":"2101.11605","n_code_links":13,"syntology":{"ran":26,"of":49,"unverified":23,"pointer_only":8}}],"papers_shown":10,"tasks":[{"task":"/task/instance-segmentation","name":"Instance Segmentation","papers":2},{"task":"/task/survival-prediction","name":"Survival Prediction","papers":2},{"task":"/task/whole-slide-images","name":"whole slide images","papers":2},{"task":"/task/anatomy","name":"Anatomy","papers":1},{"task":"/task/anomaly-detection-in-surveillance-videos","name":"Anomaly Detection In Surveillance Videos","papers":1},{"task":"/task/autonomous-driving","name":"Autonomous Driving","papers":1},{"task":"/task/classification-1","name":"Classification","papers":1},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":1},{"task":"/task/depth-estimation","name":"Depth Estimation","papers":1},{"task":"/task/dynamic-facial-expression-recognition","name":"Dynamic Facial Expression Recognition","papers":1},{"task":"/task/emotion-recognition","name":"Emotion Recognition","papers":1},{"task":"/task/facial-expression-recognition-1","name":"Facial Expression Recognition","papers":1},{"task":"/task/classification","name":"General Classification","papers":1},{"task":"/task/graph-neural-network","name":"Graph Neural Network","papers":1},{"task":"/task/image-classification","name":"Image Classification","papers":1},{"task":"/task/object-detection","name":"Object Detection","papers":1},{"task":"/task/optical-flow-estimation","name":"Optical Flow Estimation","papers":1},{"task":"/task/scene-understanding","name":"Scene Understanding","papers":1},{"task":"/task/segmentation","name":"Segmentation","papers":1},{"task":"/task/self-supervised-learning","name":"Self-Supervised Learning","papers":1}],"tasks_shown":20,"n_tasks":23,"usage_by_year":[{"year":"2021","papers":2},{"year":"2022","papers":1},{"year":"2023","papers":3},{"year":"2024","papers":2},{"year":"2025","papers":2}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/bottleneck-transformer"},"syntology_read_at":"2026-09-25T09:33:49+00:00"}