{"url":"/method/mavl","slug":"mavl","name":"MAVL","full_name":"Multiscale Attention ViT with Late fusion","full_name_withheld":false,"description_markdown":"Multiscale Attention ViT with Late fusion (MAVL) is a multi-modal network, trained with aligned image-text pairs, capable of performing targeted detection using human understandable natural language text queries. It utilizes multi-scale image features and uses deformable convolutions with late multi-modal fusion. The authors demonstrate excellent ability of MAVL as class-agnostic object detector when queried using general human understandable natural language command, such as \"all objects\", \"all entities\", etc.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Class-agnostic Object Detection with Multi-modal Transformer","paper":"/paper/multi-modal-transformers-excel-at-class","first_author":"Muhammad Maaz","n_authors":6,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/multi-modal-transformers-excel-at-class"},"source":{"url":"https://arxiv.org/abs/2111.11430v6","title":"Class-agnostic Object Detection with Multi-modal Transformer","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Computer Vision","area_id":"computer-vision","collection":"Multi-Modal Methods","url":"/methods/category/multi-modal-methods","pwc_aliases":[]}],"n_papers_tagged":4,"archive_num_papers":4,"papers_newest_first":[{"paper":"/paper/mavl-a-multilingual-audio-video-lyrics","title":"MAVL: A Multilingual Audio-Video Lyrics Dataset for Animated Song Translation","date":"2025-05-24","arxiv_id":"2505.18614","n_code_links":1,"syntology":{"ran":0,"of":19,"unverified":19,"pointer_only":0}},{"paper":null,"title":"Benchmarking Chest X-ray Diagnosis Models Across Multinational Datasets","date":"2025-05-21","arxiv_id":"2505.16027","n_code_links":0,"syntology":null},{"paper":"/paper/bridging-the-gap-between-object-and-image","title":"Bridging the Gap between Object and Image-level Representations for Open-Vocabulary Detection","date":"2022-07-07","arxiv_id":"2207.03482","n_code_links":1,"syntology":{"ran":2,"of":4,"unverified":2,"pointer_only":0}},{"paper":"/paper/multi-modal-transformers-excel-at-class","title":"Class-agnostic Object Detection with Multi-modal Transformer","date":"2021-11-22","arxiv_id":"2111.11430","n_code_links":1,"syntology":null}],"papers_shown":4,"tasks":[{"task":"/task/object","name":"Object","papers":2},{"task":"/task/benchmarking","name":"Benchmarking","papers":1},{"task":"/task/class-agnostic-object-detection","name":"Class-agnostic Object Detection","papers":1},{"task":"/task/diagnostic","name":"Diagnostic","papers":1},{"task":"/task/object-detection","name":"Object Detection","papers":1},{"task":"/task/object-proposal-generation","name":"Object Proposal Generation","papers":1},{"task":"/task/open-vocabulary-attribute-detection","name":"Open Vocabulary Attribute Detection","papers":1},{"task":"/task/open-vocabulary-object-detection","name":"Open Vocabulary Object Detection","papers":1},{"task":"/task/open-world-object-detection","name":"Open World Object Detection","papers":1},{"task":"/task/rhythm","name":"Rhythm","papers":1},{"task":"/task/translation","name":"Translation","papers":1},{"task":"/task/zero-shot-object-detection","name":"Zero-Shot Object Detection","papers":1},{"task":"/task/object-detection-1","name":"object-detection","papers":1}],"tasks_shown":13,"n_tasks":13,"usage_by_year":[{"year":"2021","papers":1},{"year":"2022","papers":1},{"year":"2025","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/mavl"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}