{"url":"/method/deformable-detr","slug":"deformable-detr","name":"Deformable DETR","full_name":"Deformable DETR","full_name_withheld":false,"description_markdown":"**Deformable DETR** is an object detection method that aims mitigates the slow convergence and high complexity issues of [DETR](https://www.paperswithcode.com/method/detr). It combines the best of the sparse spatial sampling of [deformable convolution](https://paperswithcode.com/method/deformable-convolution), and the relation modeling capability of [Transformers](https://paperswithcode.com/methods/category/transformers). Specifically, it introduces a \r\n deformable attention module, which attends to a small set of sampling locations as a pre-filter for prominent key elements out of all the feature map pixels. The module can be naturally extended to aggregating multi-scale features, without the help of [FPN](https://paperswithcode.com/method/fpn).","description_state":"present","introduced_year":null,"introduced_by":{"title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","paper":"/paper/deformable-detr-deformable-transformers-for-1","first_author":"Xizhou Zhu","n_authors":6,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/deformable-detr-deformable-transformers-for-1"},"source":{"url":"https://arxiv.org/abs/2010.04159v4","title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Vision Transformers","url":"/methods/category/vision-transformers","pwc_aliases":["vision-transformer"]},{"area":"Computer Vision","area_id":"computer-vision","collection":"Object Detection Models","url":"/methods/category/object-detection-models","pwc_aliases":[]}],"n_papers_tagged":35,"archive_num_papers":35,"papers_newest_first":[{"paper":"/paper/walnutdata-a-uav-remote-sensing-dataset-of","title":"WalnutData: A UAV Remote Sensing Dataset of Green Walnuts and Model Evaluation","date":"2025-02-27","arxiv_id":"2502.20092","n_code_links":1,"syntology":null},{"paper":"/paper/simltd-simple-supervised-and-semi-supervised","title":"SimLTD: Simple Supervised and Semi-Supervised Long-Tailed Object Detection","date":"2024-12-28","arxiv_id":"2412.20047","n_code_links":1,"syntology":null},{"paper":null,"title":"Advancing SEM Based Nano-Scale Defect Analysis in Semiconductor Manufacturing for Advanced IC Nodes","date":"2024-09-06","arxiv_id":"2409.04310","n_code_links":0,"syntology":null},{"paper":"/paper/u-decn-end-to-end-underwater-object-detection","title":"U-DECN: End-to-End Underwater Object Detection ConvNet with Improved DeNoising Training","date":"2024-08-11","arxiv_id":"2408.05780","n_code_links":1,"syntology":null},{"paper":null,"title":"Fisher-aware Quantization for DETR Detectors with Critical-category Objectives","date":"2024-07-03","arxiv_id":"2407.03442","n_code_links":0,"syntology":null},{"paper":null,"title":"Understanding differences in applying DETR to natural and medical images","date":"2024-05-27","arxiv_id":"2405.17677","n_code_links":0,"syntology":null},{"paper":null,"title":"Infrared Adversarial Car Stickers","date":"2024-05-16","arxiv_id":"2405.09924","n_code_links":0,"syntology":null},{"paper":"/paper/generative-region-language-pretraining-for","title":"Generative Region-Language Pretraining for Open-Ended Object Detection","date":"2024-03-15","arxiv_id":"2403.10191","n_code_links":1,"syntology":{"ran":8,"of":11,"unverified":3,"pointer_only":0}},{"paper":"/paper/hybrid-proposal-refiner-revisiting-detr","title":"Hybrid Proposal Refiner: Revisiting DETR Series from the Faster R-CNN Perspective","date":"2024-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"title":"KD-DETR: Knowledge Distillation for Detection Transformer with Consistent Distillation Points Sampling","date":"2024-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/mono3dvg-3d-visual-grounding-in-monocular","title":"Mono3DVG: 3D Visual Grounding in Monocular Images","date":"2023-12-13","arxiv_id":"2312.08022","n_code_links":1,"syntology":{"ran":5,"of":7,"unverified":2,"pointer_only":7}},{"paper":"/paper/towards-few-annotation-learning-for-object","title":"Towards Few-Annotation Learning for Object Detection: Are Transformer-based Models More Efficient ?","date":"2023-10-30","arxiv_id":"2310.19936","n_code_links":1,"syntology":null},{"paper":"/paper/dac-detr-divide-the-attention-layers-and","title":"DAC-DETR: Divide the Attention Layers and Conquer","date":"2023-09-21","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"title":"Selecting Learnable Training Samples is All DETRs Need in Crowded Pedestrian Detection","date":"2023-05-18","arxiv_id":"2305.10801","n_code_links":0,"syntology":null},{"paper":null,"title":"Robust Traffic Light Detection Using Salience-Sensitive Loss: Computational Framework and Evaluations","date":"2023-05-08","arxiv_id":"2305.04516","n_code_links":0,"syntology":null},{"paper":null,"title":"Continual Detection Transformer for Incremental Object Detection","date":"2023-04-06","arxiv_id":"2304.03110","n_code_links":0,"syntology":null},{"paper":"/paper/vddt-improving-vessel-detection-with","title":"VDDT: Improving Vessel Detection with Deformable Transfomer","date":"2023-03-15","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Salient Sign Detection In Safe Autonomous Driving: AI Which Reasons Over Full Visual Context","date":"2023-01-14","arxiv_id":"2301.05804","n_code_links":0,"syntology":null},{"paper":null,"title":"Open World DETR: Transformer based Open World Object Detection","date":"2022-12-06","arxiv_id":"2212.02969","n_code_links":0,"syntology":null},{"paper":"/paper/fqdet-fast-converging-query-based-detector","title":"FQDet: Fast-converging Query-based Detector","date":"2022-10-05","arxiv_id":"2210.02318","n_code_links":2,"syntology":{"ran":1,"of":6,"unverified":5,"pointer_only":3}},{"paper":null,"title":"ComplETR: Reducing the cost of annotations for object detection in dense scenes with vision transformers","date":"2022-09-13","arxiv_id":"2209.05654","n_code_links":0,"syntology":null},{"paper":"/paper/dptext-detr-towards-better-scene-text","title":"DPText-DETR: Towards Better Scene Text Detection with Dynamic Points in Transformer","date":"2022-07-10","arxiv_id":"2207.04491","n_code_links":1,"syntology":null},{"paper":"/paper/yolov7-trainable-bag-of-freebies-sets-new","title":"YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors","date":"2022-07-06","arxiv_id":"2207.02696","n_code_links":21,"syntology":{"ran":1,"of":11,"unverified":10,"pointer_only":0}},{"paper":"/paper/an-empirical-study-of-self-supervised","title":"An Empirical Study Of Self-supervised Learning Approaches For Object Detection With Transformers","date":"2022-05-11","arxiv_id":"2205.05543","n_code_links":2,"syntology":null},{"paper":"/paper/cross-domain-object-detection-with-mean","title":"MTTrans: Cross-Domain Object Detection with Mean-Teacher Transformer","date":"2022-05-03","arxiv_id":"2205.01643","n_code_links":1,"syntology":null},{"paper":null,"title":"DFAM-DETR: Deformable feature based attention mechanism DETR on slender object detection","date":"2022-04-22","arxiv_id":"2204.10667","n_code_links":0,"syntology":null},{"paper":"/paper/task-specific-attention-is-one-more-thing-you","title":"Task Specific Attention is one more thing you need for object detection","date":"2022-02-18","arxiv_id":"2202.09048","n_code_links":1,"syntology":null},{"paper":"/paper/transvod-end-to-end-video-object-detection","title":"TransVOD: End-to-End Video Object Detection with Spatial-Temporal Transformers","date":"2022-01-13","arxiv_id":"2201.05047","n_code_links":3,"syntology":{"ran":4,"of":6,"unverified":2,"pointer_only":3}},{"paper":"/paper/recurrent-glimpse-based-decoder-for-detection","title":"Recurrent Glimpse-based Decoder for Detection with Transformer","date":"2021-12-09","arxiv_id":"2112.04632","n_code_links":1,"syntology":null},{"paper":"/paper/sparse-detr-efficient-end-to-end-object-1","title":"Sparse DETR: Efficient End-to-End Object Detection with Learnable Sparsity","date":"2021-11-29","arxiv_id":"2111.14330","n_code_links":1,"syntology":{"ran":3,"of":7,"unverified":4,"pointer_only":0}}],"papers_shown":30,"tasks":[{"task":"/task/object-detection","name":"Object Detection","papers":28},{"task":"/task/object-detection-1","name":"object-detection","papers":22},{"task":"/task/object","name":"Object","papers":16},{"task":"/task/decoder","name":"Decoder","papers":8},{"task":"/task/2d-object-detection","name":"2D Object Detection","papers":3},{"task":"/task/autonomous-driving","name":"Autonomous Driving","papers":3},{"task":"/task/knowledge-distillation","name":"Knowledge Distillation","papers":3},{"task":"/task/pedestrian-detection","name":"Pedestrian Detection","papers":3},{"task":"/task/semi-supervised-object-detection","name":"Semi-Supervised Object Detection","papers":3},{"task":"/task/few-shot-object-detection","name":"Few-Shot Object Detection","papers":2},{"task":null,"name":"GPU","papers":2},{"task":"/task/instance-segmentation","name":"Instance Segmentation","papers":2},{"task":"/task/language-modeling","name":"Language Modeling","papers":2},{"task":"/task/language-modelling","name":"Language Modelling","papers":2},{"task":"/task/object-localization","name":"Object Localization","papers":2},{"task":"/task/optical-flow-estimation","name":"Optical Flow Estimation","papers":2},{"task":"/task/real-time-object-detection","name":"Real-Time Object Detection","papers":2},{"task":"/task/region-proposal","name":"Region Proposal","papers":2},{"task":"/task/segmentation","name":"Segmentation","papers":2},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":2}],"tasks_shown":20,"n_tasks":57,"usage_by_year":[{"year":"2020","papers":1},{"year":"2021","papers":6},{"year":"2022","papers":10},{"year":"2023","papers":8},{"year":"2024","papers":9},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/deformable-detr"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}