{"url":"/method/mdetr","slug":"mdetr","name":"MDETR","full_name":"MDETR","full_name_withheld":false,"description_markdown":"**MDETR** is an end-to-end modulated detector that detects objects in an image conditioned on a raw text query, like a caption or a question. It utilizes a [transformer](https://paperswithcode.com/method/transformer)-based architecture to reason jointly over text and image by fusing the two modalities at an early stage of the model. The network is pre-trained on 1.3M text-image pairs, mined from pre-existing multi-modal datasets having explicit alignment between phrases in text and objects in the image. The network is then fine-tuned on several downstream tasks such as phrase grounding, referring expression comprehension and segmentation.","description_state":"present","introduced_year":null,"introduced_by":{"title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","paper":"/paper/mdetr-modulated-detection-for-end-to-end","first_author":"Aishwarya Kamath","n_authors":6,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/mdetr-modulated-detection-for-end-to-end"},"source":{"url":"https://arxiv.org/abs/2104.12763v2","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Object Detection Models","url":"/methods/category/object-detection-models","pwc_aliases":[]}],"n_papers_tagged":13,"archive_num_papers":13,"papers_newest_first":[{"paper":"/paper/disambiguating-reference-in-visually-grounded","title":"Disambiguating Reference in Visually Grounded Dialogues through Joint Modeling of Textual and Multimodal Semantic Structures","date":"2025-05-16","arxiv_id":"2505.11726","n_code_links":1,"syntology":{"ran":2,"of":6,"unverified":4,"pointer_only":0}},{"paper":null,"title":"Seeing More with Less: Human-like Representations in Vision Models","date":"2025-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/lightmdetr-a-lightweight-approach-for-low","title":"A Lightweight Modular Framework for Low-Cost Open-Vocabulary Object Detection Training","date":"2024-08-20","arxiv_id":"2408.10787","n_code_links":1,"syntology":null},{"paper":null,"title":"ELSA: Evaluating Localization of Social Activities in Urban Streets using Open-Vocabulary Detection","date":"2024-06-03","arxiv_id":"2406.01551","n_code_links":0,"syntology":null},{"paper":"/paper/augment-the-pairs-semantics-preserving-image","title":"Augment the Pairs: Semantics-Preserving Image-Caption Pair Augmentation for Grounding-Based Vision and Language Models","date":"2023-11-05","arxiv_id":"2311.02536","n_code_links":1,"syntology":null},{"paper":"/paper/3d-aware-visual-question-answering-about-1","title":"3D-Aware Visual Question Answering about Parts, Poses and Occlusions","date":"2023-10-27","arxiv_id":"2310.17914","n_code_links":2,"syntology":{"ran":11,"of":28,"unverified":17,"pointer_only":14}},{"paper":null,"title":"Dynamic Inference With Grounding Based Vision and Language Models","date":"2023-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Dynamic MDETR: A Dynamic Multimodal Transformer Decoder for Visual Grounding","date":"2022-09-28","arxiv_id":"2209.13959","n_code_links":0,"syntology":null},{"paper":"/paper/exploring-modulated-detection-transformer-as","title":"Exploring Modulated Detection Transformer as a Tool for Action Recognition in Videos","date":"2022-09-21","arxiv_id":"2209.10126","n_code_links":1,"syntology":null},{"paper":"/paper/looking-outside-the-box-to-ground-language-in","title":"Bottom Up Top Down Detection Transformers for Language Grounding in Images and Point Clouds","date":"2021-12-16","arxiv_id":"2112.08879","n_code_links":1,"syntology":null},{"paper":null,"title":"Augmented 2D-TAN: A Two-stage Approach for Human-centric Spatio-Temporal Video Grounding","date":"2021-06-20","arxiv_id":"2106.10634","n_code_links":0,"syntology":null},{"paper":"/paper/team-ruc-aim3-technical-report-at-activitynet-1","title":"Team RUC_AIM3 Technical Report at ActivityNet 2021: Entities Object Localization","date":"2021-06-11","arxiv_id":"2106.06138","n_code_links":1,"syntology":null},{"paper":"/paper/mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","arxiv_id":"2104.12763","n_code_links":5,"syntology":{"ran":6,"of":11,"unverified":5,"pointer_only":0}}],"papers_shown":13,"tasks":[{"task":"/task/referring-expression-comprehension","name":"Referring Expression Comprehension","papers":5},{"task":"/task/phrase-grounding","name":"Phrase Grounding","papers":4},{"task":"/task/question-answering","name":"Question Answering","papers":4},{"task":"/task/referring-expression","name":"Referring Expression","papers":4},{"task":"/task/visual-question-answering","name":"Visual Question Answering (VQA)","papers":4},{"task":"/task/object-detection","name":"Object Detection","papers":3},{"task":"/task/visual-grounding","name":"Visual Grounding","papers":3},{"task":"/task/visual-question-answering-1","name":"Visual Question Answering","papers":3},{"task":"/task/object-detection-1","name":"object-detection","papers":3},{"task":"/task/action-recognition-in-videos","name":"Action Recognition","papers":2},{"task":"/task/object","name":"Object","papers":2},{"task":"/task/referring-expression-segmentation","name":"Referring Expression Segmentation","papers":2},{"task":"/task/action-detection","name":"Action Detection","papers":1},{"task":"/task/action-recognition-in-videos-2","name":"Action Recognition In Videos","papers":1},{"task":"/task/autonomous-vehicles","name":"Autonomous Vehicles","papers":1},{"task":"/task/benchmarking","name":"Benchmarking","papers":1},{"task":"/task/caption-generation","name":"Caption Generation","papers":1},{"task":"/task/computational-efficiency","name":"Computational Efficiency","papers":1},{"task":"/task/coreference-resolution","name":"Coreference Resolution","papers":1},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":1}],"tasks_shown":20,"n_tasks":34,"usage_by_year":[{"year":"2021","papers":4},{"year":"2022","papers":2},{"year":"2023","papers":3},{"year":"2024","papers":2},{"year":"2025","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/mdetr"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}