{"url":"/task/open-vocabulary-object-detection","name":"Open Vocabulary Object Detection","slug":"open-vocabulary-object-detection","description_markdown":"Open-vocabulary detection (OVD) aims to generalize beyond the limited number of base classes labeled during the training phase. The goal is to detect novel classes defined by an unbounded\r\n(open) vocabulary at inference.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":145,"papers_with_code":92,"benchmarks":4,"benchmark_tables_in_archive":4,"benchmark_tables_shown":4,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":7,"subtasks":1,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/open-vocabulary-object-detection-on-mscoco","slug":"open-vocabulary-object-detection-on-mscoco","dataset":"MSCOCO","dataset_url":"/dataset/mscoco","rows_in_archive":32,"metrics":["AP 0.5"],"first_row_in_archive_order":{"model":"Cooperative Foundational Models","paper_title":"Enhancing Novel Object Detection via Cooperative Foundational Models","paper_url":"/paper/enhancing-novel-object-detection-via","paper_date":"2023-11-19","arxiv_id":"2311.12068","code_links":[{"title":"rohit901/cooperative-foundational-models","url":"https://github.com/rohit901/cooperative-foundational-models"}],"syntology":null}},{"leaderboard":"/sota/open-vocabulary-object-detection-on-lvis-v1-0","slug":"open-vocabulary-object-detection-on-lvis-v1-0","dataset":"LVIS v1.0","dataset_url":"/dataset/lvis","rows_in_archive":28,"metrics":["AP novel-LVIS base training","AP novel-Unrestricted open-vocabulary training"],"first_row_in_archive_order":{"model":"LaMI-DETR","paper_title":"LaMI-DETR: Open-Vocabulary Detection with Language Model Instruction","paper_url":"/paper/lami-detr-open-vocabulary-detection-with","paper_date":"2024-07-16","arxiv_id":"2407.11335","code_links":[{"title":"eternaldolphin/lami-detr","url":"https://github.com/eternaldolphin/lami-detr"}],"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/open-vocabulary-object-detection-on","slug":"open-vocabulary-object-detection-on","dataset":"OpenImages-v4","dataset_url":null,"rows_in_archive":2,"metrics":["mask AP50","AP 0.5"],"first_row_in_archive_order":{"model":"Object-Centric-OVD","paper_title":"Bridging the Gap between Object and Image-level Representations for Open-Vocabulary Detection","paper_url":"/paper/bridging-the-gap-between-object-and-image","paper_date":"2022-07-07","arxiv_id":"2207.03482","code_links":[{"title":"mmaaz60/mvits_for_class_agnostic_od","url":"https://github.com/mmaaz60/mvits_for_class_agnostic_od"}],"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}}},{"leaderboard":"/sota/open-vocabulary-object-detection-on-1","slug":"open-vocabulary-object-detection-on-1","dataset":"Objects365","dataset_url":"/dataset/objects365","rows_in_archive":2,"metrics":["mask AP50"],"first_row_in_archive_order":{"model":"Object-Centric-OVD","paper_title":"Bridging the Gap between Object and Image-level Representations for Open-Vocabulary Detection","paper_url":"/paper/bridging-the-gap-between-object-and-image","paper_date":"2022-07-07","arxiv_id":"2207.03482","code_links":[{"title":"mmaaz60/mvits_for_class_agnostic_od","url":"https://github.com/mmaaz60/mvits_for_class_agnostic_od"}],"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/coco","name":"COCO (Common Objects in Context)","full_name":"Common Objects in Context","num_papers_in_archive":11922},{"url":"/dataset/lvis","name":"LVIS","full_name":"","num_papers_in_archive":551},{"url":"/dataset/objects365","name":"Objects365","full_name":"","num_papers_in_archive":161},{"url":"/dataset/mscoco","name":"MSCOCO","full_name":"","num_papers_in_archive":90},{"url":"/dataset/ovad-benchmark","name":"OVAD benchmark","full_name":"Open-Vocabulary Attribute Detection","num_papers_in_archive":14},{"url":"/dataset/description-detection-dataset","name":"Description Detection Dataset","full_name":"Description Detection Dataset","num_papers_in_archive":10},{"url":"/dataset/ovic-datasets","name":"OVIC Datasets","full_name":"Open Vocabulary Image Classification Datasets","num_papers_in_archive":1}],"subtasks":[{"url":"/task/open-vocabulary-attribute-detection","name":"Open Vocabulary Attribute Detection"}],"parent_tasks":[{"url":"/task/object-detection","name":"Object Detection"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":92,"tagged_in_all":145,"items":[{"url":"/paper/open-vocabulary-detr-with-conditional","title":"Open-Vocabulary DETR with Conditional Matching","date":"2022-03-22","arxiv_id":"2203.11876","repositories_listed":4,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":4}},{"url":"/paper/open-vocabulary-detr-with-conditional","title":"Open-Vocabulary DETR with Conditional Matching","date":"2022-03-22","arxiv_id":"2203.11876","repositories_listed":4,"syntology":{"n":8,"n_ran":3,"n_unverified":5,"n_pointer_only":4}},{"url":"/paper/zero-shot-detection-via-vision-and-language","title":"Open-vocabulary Object Detection via Vision and Language Knowledge Distillation","date":"2021-04-28","arxiv_id":"2104.13921","repositories_listed":4,"syntology":null},{"url":"/paper/zero-shot-detection-via-vision-and-language","title":"Open-vocabulary Object Detection via Vision and Language Knowledge Distillation","date":"2021-04-28","arxiv_id":"2104.13921","repositories_listed":4,"syntology":null},{"url":"/paper/seg-zero-reasoning-chain-guided-segmentation","title":"Seg-Zero: Reasoning-Chain Guided Segmentation via Cognitive Reinforcement","date":"2025-03-09","arxiv_id":"2503.06520","repositories_listed":3,"syntology":null},{"url":"/paper/yolo-world-real-time-open-vocabulary-object","title":"YOLO-World: Real-Time Open-Vocabulary Object Detection","date":"2024-01-30","arxiv_id":"2401.17270","repositories_listed":3,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/yolo-world-real-time-open-vocabulary-object","title":"YOLO-World: Real-Time Open-Vocabulary Object Detection","date":"2024-01-30","arxiv_id":"2401.17270","repositories_listed":3,"syntology":{"n":3,"n_ran":0,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/scaling-open-vocabulary-object-detection-1","title":"Scaling Open-Vocabulary Object Detection","date":"2023-06-16","arxiv_id":"2306.09683","repositories_listed":3,"syntology":null},{"url":"/paper/scaling-open-vocabulary-object-detection-1","title":"Scaling Open-Vocabulary Object Detection","date":"2023-06-16","arxiv_id":"2306.09683","repositories_listed":3,"syntology":null},{"url":"/paper/unconstrained-open-vocabulary-image","title":"Unconstrained Open Vocabulary Image Classification: Zero-Shot Transfer from Text to Image via CLIP Inversion","date":"2024-07-15","arxiv_id":"2407.11211","repositories_listed":2,"syntology":null},{"url":"/paper/is-clip-the-main-roadblock-for-fine-grained","title":"Is CLIP the main roadblock for fine-grained open-world perception?","date":"2024-04-04","arxiv_id":"2404.03539","repositories_listed":2,"syntology":{"n":17,"n_ran":15,"n_unverified":2,"n_pointer_only":17}},{"url":"/paper/is-clip-the-main-roadblock-for-fine-grained","title":"Is CLIP the main roadblock for fine-grained open-world perception?","date":"2024-04-04","arxiv_id":"2404.03539","repositories_listed":2,"syntology":{"n":17,"n_ran":15,"n_unverified":2,"n_pointer_only":17}},{"url":"/paper/real-time-transformer-based-open-vocabulary","title":"Real-time Transformer-based Open-Vocabulary Detection with Efficient Fusion Head","date":"2024-03-11","arxiv_id":"2403.06892","repositories_listed":2,"syntology":null},{"url":"/paper/real-time-transformer-based-open-vocabulary","title":"Real-time Transformer-based Open-Vocabulary Detection with Efficient Fusion Head","date":"2024-03-11","arxiv_id":"2403.06892","repositories_listed":2,"syntology":null},{"url":"/paper/cross-domain-few-shot-object-detection-via","title":"Cross-Domain Few-Shot Object Detection via Enhanced Open-Set Object Detector","date":"2024-02-05","arxiv_id":"2402.03094","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/detection-oriented-image-text-pretraining-for","title":"Region-centric Image-Language Pretraining for Open-Vocabulary Detection","date":"2023-09-29","arxiv_id":"2310.00161","repositories_listed":2,"syntology":null},{"url":"/paper/detection-oriented-image-text-pretraining-for","title":"Region-centric Image-Language Pretraining for Open-Vocabulary Detection","date":"2023-09-29","arxiv_id":"2310.00161","repositories_listed":2,"syntology":null},{"url":"/paper/improving-pseudo-labels-for-open-vocabulary","title":"Taming Self-Training for Open-Vocabulary Object Detection","date":"2023-08-11","arxiv_id":"2308.06412","repositories_listed":2,"syntology":null},{"url":"/paper/improving-pseudo-labels-for-open-vocabulary","title":"Taming Self-Training for Open-Vocabulary Object Detection","date":"2023-08-11","arxiv_id":"2308.06412","repositories_listed":2,"syntology":null},{"url":"/paper/region-aware-pretraining-for-open-vocabulary","title":"Region-Aware Pretraining for Open-Vocabulary Object Detection with Vision Transformers","date":"2023-05-11","arxiv_id":"2305.07011","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/region-aware-pretraining-for-open-vocabulary","title":"Region-Aware Pretraining for Open-Vocabulary Object Detection with Vision Transformers","date":"2023-05-11","arxiv_id":"2305.07011","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/x-paste-revisit-copy-paste-at-scale-with-clip","title":"X-Paste: Revisiting Scalable Copy-Paste for Instance Segmentation using CLIP and StableDiffusion","date":"2022-12-07","arxiv_id":"2212.03863","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/pointclip-v2-adapting-clip-for-powerful-3d","title":"PointCLIP V2: Prompting CLIP and GPT for Powerful 3D Open-world Learning","date":"2022-11-21","arxiv_id":"2211.11682","repositories_listed":2,"syntology":{"n":12,"n_ran":5,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/open-vocabulary-object-detection-with","title":"Open Vocabulary Object Detection with Proposal Mining and Prediction Equalization","date":"2022-06-22","arxiv_id":"2206.11134","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/open-vocabulary-object-detection-with","title":"Open Vocabulary Object Detection with Proposal Mining and Prediction Equalization","date":"2022-06-22","arxiv_id":"2206.11134","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/simple-open-vocabulary-object-detection-with","title":"Simple Open-Vocabulary Object Detection with Vision Transformers","date":"2022-05-12","arxiv_id":"2205.06230","repositories_listed":2,"syntology":null},{"url":"/paper/simple-open-vocabulary-object-detection-with","title":"Simple Open-Vocabulary Object Detection with Vision Transformers","date":"2022-05-12","arxiv_id":"2205.06230","repositories_listed":2,"syntology":null},{"url":"/paper/pointclip-point-cloud-understanding-by-clip","title":"PointCLIP: Point Cloud Understanding by CLIP","date":"2021-12-04","arxiv_id":"2112.02413","repositories_listed":2,"syntology":null},{"url":"/paper/fg-clip-fine-grained-visual-and-textual","title":"FG-CLIP: Fine-Grained Visual and Textual Alignment","date":"2025-05-08","arxiv_id":"2505.05071","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/fg-clip-fine-grained-visual-and-textual","title":"FG-CLIP: Fine-Grained Visual and Textual Alignment","date":"2025-05-08","arxiv_id":"2505.05071","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}}],"syntology_records":15,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}