{"url":"/task/zero-shot-object-detection","name":"Zero-Shot Object Detection","slug":"zero-shot-object-detection","description_markdown":"Zero-shot object detection (ZSD) is the task of object detection where no visual training data is available for some of the target object classes.\r\n\r\n<span style=\"color:grey; opacity: 0.6\">( Image credit: [Zero-Shot Object Detection: Learning to Simultaneously Recognize and Localize Novel Concepts](https://github.com/salman-h-khan/ZSD_Release) )</span>","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":57,"papers_with_code":39,"benchmarks":7,"benchmark_tables_in_archive":7,"benchmark_tables_shown":7,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":6,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/zero-shot-object-detection-on-lvis-v1-0","slug":"zero-shot-object-detection-on-lvis-v1-0","dataset":"LVIS v1.0 minival","dataset_url":"/dataset/lvis","rows_in_archive":11,"metrics":["AP"],"first_row_in_archive_order":{"model":"CP-DETR-Pro(without LVIS data)","paper_title":"CP-DETR: Concept Prompt Guide DETR Toward Stronger Universal Object Detection","paper_url":"/paper/cp-detr-concept-prompt-guide-detr-toward","paper_date":"2024-12-13","arxiv_id":"2412.09799","code_links":[],"syntology":null}},{"leaderboard":"/sota/zero-shot-object-detection-on-lvis-v1-0-val","slug":"zero-shot-object-detection-on-lvis-v1-0-val","dataset":"LVIS v1.0 val","dataset_url":"/dataset/lvis","rows_in_archive":9,"metrics":["AP"],"first_row_in_archive_order":{"model":"CP-DETR-Pro(without LVIS data)","paper_title":"CP-DETR: Concept Prompt Guide DETR Toward Stronger Universal Object Detection","paper_url":"/paper/cp-detr-concept-prompt-guide-detr-toward","paper_date":"2024-12-13","arxiv_id":"2412.09799","code_links":[],"syntology":null}},{"leaderboard":"/sota/zero-shot-object-detection-on-ms-coco","slug":"zero-shot-object-detection-on-ms-coco","dataset":"MS-COCO","dataset_url":"/dataset/coco","rows_in_archive":9,"metrics":["mAP","Recall"],"first_row_in_archive_order":{"model":"UniFa","paper_title":"UniFa: A unified feature hallucination framework for any-shot object detection","paper_url":"/paper/unifa-a-unified-feature-hallucination","paper_date":"2025-03-01","arxiv_id":null,"code_links":[],"syntology":null}},{"leaderboard":"/sota/zero-shot-object-detection-on-mscoco","slug":"zero-shot-object-detection-on-mscoco","dataset":"MSCOCO","dataset_url":"/dataset/mscoco","rows_in_archive":7,"metrics":["AP"],"first_row_in_archive_order":{"model":"Grounding DINO 1.6 Pro (without COCO data)","paper_title":"Grounding DINO 1.5: Advance the \"Edge\" of Open-Set Object Detection","paper_url":"/paper/grounding-dino-1-5-advance-the-edge-of-open","paper_date":"2024-05-16","arxiv_id":"2405.10300","code_links":[{"title":"mit-han-lab/efficientvit","url":"https://github.com/mit-han-lab/efficientvit"},{"title":"idea-research/grounded-sam-2","url":"https://github.com/idea-research/grounded-sam-2"},{"title":"idea-research/grounding-dino-1.5-api","url":"https://github.com/idea-research/grounding-dino-1.5-api"}],"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/zero-shot-object-detection-on-pascal-voc-07","slug":"zero-shot-object-detection-on-pascal-voc-07","dataset":"PASCAL VOC'07","dataset_url":null,"rows_in_archive":7,"metrics":["mAP"],"first_row_in_archive_order":{"model":"SeeDS","paper_title":"SeeDS: Semantic Separable Diffusion Synthesizer for Zero-shot Food Detection","paper_url":"/paper/seeds-semantic-separable-diffusion","paper_date":"2023-10-07","arxiv_id":"2310.04689","code_links":[{"title":"lancezpf/seeds","url":"https://github.com/lancezpf/seeds"}],"syntology":null}},{"leaderboard":"/sota/zero-shot-object-detection-on-odinw","slug":"zero-shot-object-detection-on-odinw","dataset":"ODinW","dataset_url":null,"rows_in_archive":5,"metrics":["Average Score"],"first_row_in_archive_order":{"model":"CP-DETR-L Swin-L","paper_title":"CP-DETR: Concept Prompt Guide DETR Toward Stronger Universal Object Detection","paper_url":"/paper/cp-detr-concept-prompt-guide-detr-toward","paper_date":"2024-12-13","arxiv_id":"2412.09799","code_links":[],"syntology":null}},{"leaderboard":"/sota/zero-shot-object-detection-on-imagenet","slug":"zero-shot-object-detection-on-imagenet","dataset":"ImageNet Detection","dataset_url":null,"rows_in_archive":1,"metrics":["mAP"],"first_row_in_archive_order":{"model":"SUZOD","paper_title":"Synthesizing the Unseen for Zero-shot Object Detection","paper_url":"/paper/synthesizing-the-unseen-for-zero-shot-object","paper_date":"2020-10-19","arxiv_id":"2010.09425","code_links":[{"title":"nasir6/zero_shot_detection","url":"https://github.com/nasir6/zero_shot_detection"},{"title":"witnessai/Awesome-Zero-Shot-Object-Detection","url":"https://github.com/witnessai/Awesome-Zero-Shot-Object-Detection"}],"syntology":null}}],"datasets":[{"url":"/dataset/coco","name":"COCO (Common Objects in Context)","full_name":"Common Objects in Context","num_papers_in_archive":11922},{"url":"/dataset/lvis","name":"LVIS","full_name":"","num_papers_in_archive":551},{"url":"/dataset/pascal-voc-2007","name":"PASCAL VOC 2007","full_name":"PASCAL VOC 2007","num_papers_in_archive":126},{"url":"/dataset/mscoco","name":"MSCOCO","full_name":"","num_papers_in_archive":90},{"url":"/dataset/elevater","name":"ELEVATER","full_name":"Evaluation of Language-augmented Visual Task-level Transfer","num_papers_in_archive":25},{"url":"/dataset/rf100","name":"RF100","full_name":"Roboflow 100","num_papers_in_archive":5}],"subtasks":[],"parent_tasks":[{"url":"/task/object-detection","name":"Object Detection"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":39,"tagged_in_all":57,"items":[{"url":"/paper/grounding-dino-marrying-dino-with-grounded","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","date":"2023-03-09","arxiv_id":"2303.05499","repositories_listed":10,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/elevater-a-benchmark-and-toolkit-for","title":"ELEVATER: A Benchmark and Toolkit for Evaluating Language-Augmented Visual Models","date":"2022-04-19","arxiv_id":"2204.08790","repositories_listed":9,"syntology":{"n":20,"n_ran":4,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/learning-open-world-object-proposals-without","title":"Learning Open-World Object Proposals without Learning to Classify","date":"2021-08-15","arxiv_id":"2108.06753","repositories_listed":6,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/zero-shot-detection-via-vision-and-language","title":"Open-vocabulary Object Detection via Vision and Language Knowledge Distillation","date":"2021-04-28","arxiv_id":"2104.13921","repositories_listed":4,"syntology":null},{"url":"/paper/zero-shot-instance-segmentation","title":"Zero-Shot Instance Segmentation","date":"2021-04-14","arxiv_id":"2104.06601","repositories_listed":4,"syntology":null},{"url":"/paper/visionreasoner-unified-visual-perception-and","title":"VisionReasoner: Unified Visual Perception and Reasoning via Reinforcement Learning","date":"2025-05-17","arxiv_id":"2505.12081","repositories_listed":3,"syntology":{"n":15,"n_ran":1,"n_unverified":14,"n_pointer_only":0}},{"url":"/paper/grounding-dino-1-5-advance-the-edge-of-open","title":"Grounding DINO 1.5: Advance the \"Edge\" of Open-Set Object Detection","date":"2024-05-16","arxiv_id":"2405.10300","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/yolo-world-real-time-open-vocabulary-object","title":"YOLO-World: Real-Time Open-Vocabulary Object Detection","date":"2024-01-30","arxiv_id":"2401.17270","repositories_listed":3,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/scaling-open-vocabulary-object-detection-1","title":"Scaling Open-Vocabulary Object Detection","date":"2023-06-16","arxiv_id":"2306.09683","repositories_listed":3,"syntology":null},{"url":"/paper/grounded-language-image-pre-training","title":"Grounded Language-Image Pre-training","date":"2021-12-07","arxiv_id":"2112.03857","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/polarity-loss-for-zero-shot-object-detection","title":"Polarity Loss for Zero-shot Object Detection","date":"2018-11-22","arxiv_id":"1811.08982","repositories_listed":3,"syntology":null},{"url":"/paper/t-rex2-towards-generic-object-detection-via","title":"T-Rex2: Towards Generic Object Detection via Text-Visual Prompt Synergy","date":"2024-03-21","arxiv_id":"2403.14610","repositories_listed":2,"syntology":null},{"url":"/paper/real-time-transformer-based-open-vocabulary","title":"Real-time Transformer-based Open-Vocabulary Detection with Efficient Fusion Head","date":"2024-03-11","arxiv_id":"2403.06892","repositories_listed":2,"syntology":null},{"url":"/paper/efficient-feature-distillation-for-zero-shot","title":"Efficient Feature Distillation for Zero-shot Annotation Object Detection","date":"2023-03-21","arxiv_id":"2303.12145","repositories_listed":2,"syntology":null},{"url":"/paper/synthesizing-the-unseen-for-zero-shot-object","title":"Synthesizing the Unseen for Zero-shot Object Detection","date":"2020-10-19","arxiv_id":"2010.09425","repositories_listed":2,"syntology":null},{"url":"/paper/zero-shot-object-detection-by-hybrid-region","title":"Zero-Shot Object Detection by Hybrid Region Embedding","date":"2018-05-16","arxiv_id":"1805.06157","repositories_listed":2,"syntology":null},{"url":"/paper/towards-a-multi-agent-vision-language-system","title":"Towards a Multi-Agent Vision-Language System for Zero-Shot Novel Hazardous Object Detection for Autonomous Driving Safety","date":"2025-04-18","arxiv_id":"2504.13399","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/langgas-introducing-language-in-selective","title":"LangGas: Introducing Language in Selective Zero-Shot Background Subtraction for Semi-Transparent Gas Leak Detection with a New Dataset","date":"2025-03-04","arxiv_id":"2503.02910","repositories_listed":1,"syntology":null},{"url":"/paper/no-annotations-for-object-detection-in-art","title":"No Annotations for Object Detection in Art through Stable Diffusion","date":"2024-12-09","arxiv_id":"2412.06286","repositories_listed":1,"syntology":null},{"url":"/paper/gaussian-splatting-under-attack-investigating","title":"Gaussian Splatting Under Attack: Investigating Adversarial Noise in 3D Objects","date":"2024-12-03","arxiv_id":"2412.02803","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_unverified":2,"n_pointer_only":2}},{"url":"/paper/dino-x-a-unified-vision-model-for-open-world","title":"DINO-X: A Unified Vision Model for Open-World Object Detection and Understanding","date":"2024-11-21","arxiv_id":"2411.14347","repositories_listed":1,"syntology":null},{"url":"/paper/ov-dino-unified-open-vocabulary-detection","title":"OV-DINO: Unified Open-Vocabulary Detection with Language-Aware Selective Fusion","date":"2024-07-10","arxiv_id":"2407.07844","repositories_listed":1,"syntology":null},{"url":"/paper/ov-dquo-open-vocabulary-detr-with-denoising","title":"OV-DQUO: Open-Vocabulary DETR with Denoising Text Query Training and Open-World Unknown Objects Supervision","date":"2024-05-28","arxiv_id":"2405.17913","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/dettoolchain-a-new-prompting-paradigm-to","title":"DetToolChain: A New Prompting Paradigm to Unleash Detection Ability of MLLM","date":"2024-03-19","arxiv_id":"2403.12488","repositories_listed":1,"syntology":null},{"url":"/paper/synthesizing-knowledge-enhanced-features-for","title":"Synthesizing Knowledge-enhanced Features for Real-world Zero-shot Food Detection","date":"2024-02-14","arxiv_id":"2402.09242","repositories_listed":1,"syntology":null},{"url":"/paper/seeds-semantic-separable-diffusion","title":"SeeDS: Semantic Separable Diffusion Synthesizer for Zero-shot Food Detection","date":"2023-10-07","arxiv_id":"2310.04689","repositories_listed":1,"syntology":null},{"url":"/paper/villa-fine-grained-vision-language","title":"ViLLA: Fine-Grained Vision-Language Representation Learning from Real-World Data","date":"2023-08-22","arxiv_id":"2308.11194","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/multi-modal-queried-object-detection-in-the","title":"Multi-modal Queried Object Detection in the Wild","date":"2023-05-30","arxiv_id":"2305.18980","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/dounseen-zero-shot-object-detection-for","title":"DoUnseen: Tuning-Free Class-Adaptive Object Detection of Unseen Objects for Robotic Grasping","date":"2023-04-06","arxiv_id":"2304.02833","repositories_listed":1,"syntology":null},{"url":"/paper/zbs-zero-shot-background-subtraction-via","title":"ZBS: Zero-shot Background Subtraction via Instance-level Background Modeling and Foreground Selection","date":"2023-03-26","arxiv_id":"2303.14679","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_unverified":4,"n_pointer_only":0}}],"syntology_records":13,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}