{"url":"/task/open-vocabulary-image-classification","name":"Open Vocabulary Image Classification","slug":"open-vocabulary-image-classification","description_markdown":"Given an image and nothing else, i.e. no prompts or candidate labels, the task is to generate an accurate textual fine-grained classification label from the entire pool of simple and compound nouns in the English language.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":7,"papers_with_code":6,"benchmarks":4,"benchmark_tables_in_archive":4,"benchmark_tables_shown":4,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":1,"subtasks":0,"parent_tasks":1},"benchmarks":[{"leaderboard":"/sota/open-vocabulary-image-classification-on-ovic","slug":"open-vocabulary-image-classification-on-ovic","dataset":"OVIC Datasets (World-H)","dataset_url":"/dataset/ovic-datasets","rows_in_archive":3,"metrics":["Prediction Score (mean of 3)","Overall Score","Prediction Score","Top 1 Accuracy"],"first_row_in_archive_order":{"model":"SigLIP SO/14 + PrefixedIter Decoder (FT2)","paper_title":"Unconstrained Open Vocabulary Image Classification: Zero-Shot Transfer from Text to Image via CLIP Inversion","paper_url":"/paper/unconstrained-open-vocabulary-image","paper_date":"2024-07-15","arxiv_id":"2407.11211","code_links":[{"title":"pallgeuer/novic","url":"https://github.com/pallgeuer/novic"},{"title":"pallgeuer/object_noun_dictionary","url":"https://github.com/pallgeuer/object_noun_dictionary"}],"syntology":null}},{"leaderboard":"/sota/open-vocabulary-image-classification-on-ovic-1","slug":"open-vocabulary-image-classification-on-ovic-1","dataset":"OVIC Datasets (Wiki-H)","dataset_url":"/dataset/ovic-datasets","rows_in_archive":3,"metrics":["Overall Score","Prediction Score","Top 1 Accuracy","Prediction Score (mean of 3)"],"first_row_in_archive_order":{"model":"DFN-5B H/14-378 + PrefixedIter Decoder (FT2)","paper_title":"Unconstrained Open Vocabulary Image Classification: Zero-Shot Transfer from Text to Image via CLIP Inversion","paper_url":"/paper/unconstrained-open-vocabulary-image","paper_date":"2024-07-15","arxiv_id":"2407.11211","code_links":[{"title":"pallgeuer/novic","url":"https://github.com/pallgeuer/novic"},{"title":"pallgeuer/object_noun_dictionary","url":"https://github.com/pallgeuer/object_noun_dictionary"}],"syntology":null}},{"leaderboard":"/sota/open-vocabulary-image-classification-on-ovic-2","slug":"open-vocabulary-image-classification-on-ovic-2","dataset":"OVIC Datasets (Val3K)","dataset_url":"/dataset/ovic-datasets","rows_in_archive":2,"metrics":["Prediction Score (mean of 3)","Top 1 Accuracy (mean of 3)"],"first_row_in_archive_order":{"model":"SigLIP B/16 + PrefixedIter Decoder (FT6)","paper_title":"Unconstrained Open Vocabulary Image Classification: Zero-Shot Transfer from Text to Image via CLIP Inversion","paper_url":"/paper/unconstrained-open-vocabulary-image","paper_date":"2024-07-15","arxiv_id":"2407.11211","code_links":[{"title":"pallgeuer/novic","url":"https://github.com/pallgeuer/novic"},{"title":"pallgeuer/object_noun_dictionary","url":"https://github.com/pallgeuer/object_noun_dictionary"}],"syntology":null}},{"leaderboard":"/sota/open-vocabulary-image-classification-on-ovic-3","slug":"open-vocabulary-image-classification-on-ovic-3","dataset":"OVIC Datasets (Wiki-L)","dataset_url":"/dataset/ovic-datasets","rows_in_archive":2,"metrics":["Prediction Score (mean of 3)"],"first_row_in_archive_order":{"model":"DFN-5B H/14-378 + PrefixedIter Decoder (FT2)","paper_title":"Unconstrained Open Vocabulary Image Classification: Zero-Shot Transfer from Text to Image via CLIP Inversion","paper_url":"/paper/unconstrained-open-vocabulary-image","paper_date":"2024-07-15","arxiv_id":"2407.11211","code_links":[{"title":"pallgeuer/novic","url":"https://github.com/pallgeuer/novic"},{"title":"pallgeuer/object_noun_dictionary","url":"https://github.com/pallgeuer/object_noun_dictionary"}],"syntology":null}}],"datasets":[{"url":"/dataset/ovic-datasets","name":"OVIC Datasets","full_name":"Open Vocabulary Image Classification Datasets","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[{"url":"/task/zero-shot-image-classification","name":"Zero-Shot Image Classification"}],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":6,"of":6,"tagged_in_all":7,"items":[{"url":"/paper/zero-shot-detection-via-vision-and-language","title":"Open-vocabulary Object Detection via Vision and Language Knowledge Distillation","date":"2021-04-28","arxiv_id":"2104.13921","repositories_listed":4,"syntology":null},{"url":"/paper/unconstrained-open-vocabulary-image","title":"Unconstrained Open Vocabulary Image Classification: Zero-Shot Transfer from Text to Image via CLIP Inversion","date":"2024-07-15","arxiv_id":"2407.11211","repositories_listed":2,"syntology":null},{"url":"/paper/2112-14757","title":"A Simple Baseline for Open-Vocabulary Semantic Segmentation with Pre-trained Vision-language Model","date":"2021-12-29","arxiv_id":"2112.14757","repositories_listed":2,"syntology":null},{"url":"/paper/anytime-continual-learning-for-open","title":"Anytime Continual Learning for Open Vocabulary Classification","date":"2024-09-13","arxiv_id":"2409.08518","repositories_listed":1,"syntology":null},{"url":"/paper/fusionbench-a-comprehensive-benchmark-of-deep","title":"FusionBench: A Comprehensive Benchmark of Deep Model Fusion","date":"2024-06-05","arxiv_id":"2406.03280","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/continual-learning-in-open-vocabulary","title":"Continual Learning in Open-vocabulary Classification with Complementary Memory Systems","date":"2023-07-04","arxiv_id":"2307.01430","repositories_listed":1,"syntology":null}],"syntology_records":1,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}