{"url":"/task/open-vocabulary-attribute-detection","name":"Open Vocabulary Attribute Detection","slug":"open-vocabulary-attribute-detection","description_markdown":"Open-Vocabulary Attribute Detection (OVAD) is a task that aims to detect and recognize an open set of objects and their associated attributes in an image. The objects and attributes are defined by text queries during inference, without prior knowledge of the tested classes during training.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":14,"papers_with_code":12,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/open-vocabulary-attribute-detection-on-ovad-1","slug":"open-vocabulary-attribute-detection-on-ovad-1","dataset":"OVAD-Box benchmark","dataset_url":"/dataset/ovad-benchmark","rows_in_archive":7,"metrics":["mean average precision"],"first_row_in_archive_order":{"model":"X-VLM","paper_title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","paper_url":"/paper/multi-grained-vision-language-pre-training","paper_date":"2021-11-16","arxiv_id":"2111.08276","code_links":[{"title":"zengyan-97/x-vlm","url":"https://github.com/zengyan-97/x-vlm"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/open-vocabulary-attribute-detection-on-ovad","slug":"open-vocabulary-attribute-detection-on-ovad","dataset":"OVAD benchmark","dataset_url":"/dataset/ovad-benchmark","rows_in_archive":5,"metrics":["mean average precision"],"first_row_in_archive_order":{"model":"OvarNet (ViT-B16)","paper_title":"OvarNet: Towards Open-vocabulary Object Attribute Recognition","paper_url":"/paper/ovarnet-towards-open-vocabulary-object","paper_date":"2023-01-23","arxiv_id":"2301.09506","code_links":[{"title":"KyanChen/OvarNet","url":"https://github.com/KyanChen/OvarNet"}],"syntology":null}}],"datasets":[{"url":"/dataset/ovad-benchmark","name":"OVAD benchmark","full_name":"Open-Vocabulary Attribute Detection","num_papers_in_archive":14}],"subtasks":[],"parent_tasks":[{"url":"/task/open-vocabulary-object-detection","name":"Open Vocabulary Object Detection"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":12,"of":12,"tagged_in_all":14,"items":[{"url":"/paper/learning-transferable-visual-models-from","title":"Learning Transferable Visual Models From Natural Language Supervision","date":"2021-02-26","arxiv_id":"2103.00020","repositories_listed":82,"syntology":{"n":20,"n_ran":16,"n_unverified":4,"n_pointer_only":16}},{"url":"/paper/blip-2-bootstrapping-language-image-pre","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","date":"2023-01-30","arxiv_id":"2301.12597","repositories_listed":17,"syntology":{"n":8,"n_ran":4,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/blip-bootstrapping-language-image-pre","title":"BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation","date":"2022-01-28","arxiv_id":"2201.12086","repositories_listed":9,"syntology":null},{"url":"/paper/align-before-fuse-vision-and-language","title":"Align before Fuse: Vision and Language Representation Learning with Momentum Distillation","date":"2021-07-16","arxiv_id":"2107.07651","repositories_listed":6,"syntology":{"n":5,"n_ran":3,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/reproducible-scaling-laws-for-contrastive","title":"Reproducible scaling laws for contrastive language-image learning","date":"2022-12-14","arxiv_id":"2212.07143","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/open-ended-vqa-benchmarking-of-vision","title":"Open-ended VQA benchmarking of Vision-Language models by exploiting Classification datasets and their semantic hierarchy","date":"2024-02-11","arxiv_id":"2402.07270","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/ovarnet-towards-open-vocabulary-object","title":"OvarNet: Towards Open-vocabulary Object Attribute Recognition","date":"2023-01-23","arxiv_id":"2301.09506","repositories_listed":1,"syntology":null},{"url":"/paper/open-vocabulary-attribute-detection","title":"Open-vocabulary Attribute Detection","date":"2022-11-23","arxiv_id":"2211.12914","repositories_listed":1,"syntology":{"n":10,"n_ran":3,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/bridging-the-gap-between-object-and-image","title":"Bridging the Gap between Object and Image-level Representations for Open-Vocabulary Detection","date":"2022-07-07","arxiv_id":"2207.03482","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/localized-vision-language-matching-for-open","title":"Localized Vision-Language Matching for Open-vocabulary Object Detection","date":"2022-05-12","arxiv_id":"2205.06160","repositories_listed":1,"syntology":null},{"url":"/paper/multi-grained-vision-language-pre-training","title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts","date":"2021-11-16","arxiv_id":"2111.08276","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/open-vocabulary-object-detection-using","title":"Open-Vocabulary Object Detection Using Captions","date":"2020-11-20","arxiv_id":"2011.10678","repositories_listed":1,"syntology":null}],"syntology_records":8,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}