{"url":"/task/zero-shot-image-classification","name":"Zero-Shot Image Classification","slug":"zero-shot-image-classification","description_markdown":"Zero-shot image classification is a technique in computer vision where a model can classify images into categories that were not present during training. This is achieved by leveraging semantic information about the categories, such as textual descriptions or relationships between classes.","categories":[{"name":"Computer Vision","url":"/area/computer-vision"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":111,"papers_with_code":64,"benchmarks":3,"benchmark_tables_in_archive":3,"benchmark_tables_shown":3,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":6,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/zero-shot-image-classification-on-country211","slug":"zero-shot-image-classification-on-country211","dataset":"Country211","dataset_url":"/dataset/country211","rows_in_archive":1,"metrics":["Top-1 accuracy"],"first_row_in_archive_order":{"model":"OpenClip H/14 (34B)(Laion2B)","paper_title":"Reproducible scaling laws for contrastive language-image learning","paper_url":"/paper/reproducible-scaling-laws-for-contrastive","paper_date":"2022-12-14","arxiv_id":"2212.07143","code_links":[{"title":"mlfoundations/open_clip","url":"https://github.com/mlfoundations/open_clip"},{"title":"laion-ai/scaling-laws-openclip","url":"https://github.com/laion-ai/scaling-laws-openclip"},{"title":"shkarupa-alex/tfclip","url":"https://github.com/shkarupa-alex/tfclip"},{"title":"nahidalam/open_clip","url":"https://github.com/nahidalam/open_clip"},{"title":"eify/open_clip","url":"https://github.com/eify/open_clip"}],"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}}},{"leaderboard":"/sota/zero-shot-image-classification-on-icinw","slug":"zero-shot-image-classification-on-icinw","dataset":"ICinW","dataset_url":null,"rows_in_archive":1,"metrics":["Average Score"],"first_row_in_archive_order":{"model":"CLIP (ViT B-32)","paper_title":"ELEVATER: A Benchmark and Toolkit for Evaluating Language-Augmented Visual Models","paper_url":"/paper/elevater-a-benchmark-and-toolkit-for","paper_date":"2022-04-19","arxiv_id":"2204.08790","code_links":[{"title":"microsoft/GLIP","url":"https://github.com/microsoft/GLIP"},{"title":"computer-vision-in-the-wild/cvinw_readings","url":"https://github.com/computer-vision-in-the-wild/cvinw_readings"},{"title":"microsoft/esvit","url":"https://github.com/microsoft/esvit"},{"title":"microsoft/unicl","url":"https://github.com/microsoft/unicl"},{"title":"eric-ai-lab/pevit","url":"https://github.com/eric-ai-lab/pevit"},{"title":"Computer-Vision-in-the-Wild/Elevater_Toolkit_IC","url":"https://github.com/Computer-Vision-in-the-Wild/Elevater_Toolkit_IC"},{"title":"sincerass/mvlpt","url":"https://github.com/sincerass/mvlpt"},{"title":"microsoft/klite","url":"https://github.com/microsoft/klite"},{"title":"rsCPSyEu/ovd_cod","url":"https://github.com/rsCPSyEu/ovd_cod"}],"syntology":{"n":20,"n_ran":4,"n_unverified":16,"n_pointer_only":0}}},{"leaderboard":"/sota/zero-shot-image-classification-on-odinw","slug":"zero-shot-image-classification-on-odinw","dataset":"ODinW","dataset_url":null,"rows_in_archive":1,"metrics":["Average Score"],"first_row_in_archive_order":{"model":"GLIP (Tiny A)","paper_title":"ELEVATER: A Benchmark and Toolkit for Evaluating Language-Augmented Visual Models","paper_url":"/paper/elevater-a-benchmark-and-toolkit-for","paper_date":"2022-04-19","arxiv_id":"2204.08790","code_links":[{"title":"microsoft/GLIP","url":"https://github.com/microsoft/GLIP"},{"title":"computer-vision-in-the-wild/cvinw_readings","url":"https://github.com/computer-vision-in-the-wild/cvinw_readings"},{"title":"microsoft/esvit","url":"https://github.com/microsoft/esvit"},{"title":"microsoft/unicl","url":"https://github.com/microsoft/unicl"},{"title":"eric-ai-lab/pevit","url":"https://github.com/eric-ai-lab/pevit"},{"title":"Computer-Vision-in-the-Wild/Elevater_Toolkit_IC","url":"https://github.com/Computer-Vision-in-the-Wild/Elevater_Toolkit_IC"},{"title":"sincerass/mvlpt","url":"https://github.com/sincerass/mvlpt"},{"title":"microsoft/klite","url":"https://github.com/microsoft/klite"},{"title":"rsCPSyEu/ovd_cod","url":"https://github.com/rsCPSyEu/ovd_cod"}],"syntology":{"n":20,"n_ran":4,"n_unverified":16,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/quilt-1m","name":"QUILT-1M","full_name":"","num_papers_in_archive":17},{"url":"/dataset/country211","name":"Country211","full_name":"Country211","num_papers_in_archive":3},{"url":"/dataset/mineralimage5k","name":"MineralImage5k","full_name":"Benchmark for 5k raw mineral species recognition","num_papers_in_archive":1},{"url":"/dataset/ovic-datasets","name":"OVIC Datasets","full_name":"Open Vocabulary Image Classification Datasets","num_papers_in_archive":1},{"url":"/dataset/taxabench-8k","name":"TaxaBench-8k","full_name":"","num_papers_in_archive":1},{"url":"/dataset/corn-seeds-dataset","name":"Corn Seeds Dataset","full_name":"","num_papers_in_archive":0}],"subtasks":[{"url":"/task/open-vocabulary-image-classification","name":"Open Vocabulary Image Classification"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":64,"tagged_in_all":111,"items":[{"url":"/paper/elevater-a-benchmark-and-toolkit-for","title":"ELEVATER: A Benchmark and Toolkit for Evaluating Language-Augmented Visual Models","date":"2022-04-19","arxiv_id":"2204.08790","repositories_listed":9,"syntology":{"n":20,"n_ran":4,"n_unverified":16,"n_pointer_only":0}},{"url":"/paper/reproducible-scaling-laws-for-contrastive","title":"Reproducible scaling laws for contrastive language-image learning","date":"2022-12-14","arxiv_id":"2212.07143","repositories_listed":5,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":3}},{"url":"/paper/lit-zero-shot-transfer-with-locked-image-text","title":"LiT: Zero-Shot Transfer with Locked-image text Tuning","date":"2021-11-15","arxiv_id":"2111.07991","repositories_listed":5,"syntology":null},{"url":"/paper/scaling-up-visual-and-vision-language","title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision","date":"2021-02-11","arxiv_id":"2102.05918","repositories_listed":5,"syntology":{"n":10,"n_ran":8,"n_unverified":2,"n_pointer_only":9}},{"url":"/paper/zero-shot-detection-via-vision-and-language","title":"Open-vocabulary Object Detection via Vision and Language Knowledge Distillation","date":"2021-04-28","arxiv_id":"2104.13921","repositories_listed":4,"syntology":null},{"url":"/paper/what-does-a-platypus-look-like-generating","title":"What does a platypus look like? Generating customized prompts for zero-shot image classification","date":"2022-09-07","arxiv_id":"2209.03320","repositories_listed":3,"syntology":null},{"url":"/paper/unconstrained-open-vocabulary-image","title":"Unconstrained Open Vocabulary Image Classification: Zero-Shot Transfer from Text to Image via CLIP Inversion","date":"2024-07-15","arxiv_id":"2407.11211","repositories_listed":2,"syntology":null},{"url":"/paper/sparse-concept-bottleneck-models-gumbel","title":"Sparse Concept Bottleneck Models: Gumbel Tricks in Contrastive Learning","date":"2024-04-04","arxiv_id":"2404.03323","repositories_listed":2,"syntology":null},{"url":"/paper/contrasting-intra-modal-and-ranking-cross","title":"Contrasting Intra-Modal and Ranking Cross-Modal Hard Negatives to Enhance Visio-Linguistic Compositional Understanding","date":"2023-06-15","arxiv_id":"2306.08832","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/altclip-altering-the-language-encoder-in-clip","title":"AltCLIP: Altering the Language Encoder in CLIP for Extended Language Capabilities","date":"2022-11-12","arxiv_id":"2211.06679","repositories_listed":2,"syntology":{"n":11,"n_ran":2,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/duet-cross-modal-semantic-grounding-for","title":"DUET: Cross-modal Semantic Grounding for Contrastive Zero-shot Learning","date":"2022-07-04","arxiv_id":"2207.01328","repositories_listed":2,"syntology":null},{"url":"/paper/2112-14757","title":"A Simple Baseline for Open-Vocabulary Semantic Segmentation with Pre-trained Vision-language Model","date":"2021-12-29","arxiv_id":"2112.14757","repositories_listed":2,"syntology":null},{"url":"/paper/lrsclip-a-vision-language-foundation-model","title":"LRSCLIP: A Vision-Language Foundation Model for Aligning Remote Sensing Image with Longer Text","date":"2025-03-25","arxiv_id":"2503.19311","repositories_listed":1,"syntology":null},{"url":"/paper/cross-the-gap-exposing-the-intra-modal","title":"Cross the Gap: Exposing the Intra-modal Misalignment in CLIP via Modality Inversion","date":"2025-02-06","arxiv_id":"2502.04263","repositories_listed":1,"syntology":null},{"url":"/paper/kpl-training-free-medical-knowledge-mining-of","title":"KPL: Training-Free Medical Knowledge Mining of Vision-Language Models","date":"2025-01-20","arxiv_id":"2501.11231","repositories_listed":1,"syntology":null},{"url":"/paper/post-hoc-probabilistic-vision-language-models","title":"Post-hoc Probabilistic Vision-Language Models","date":"2024-12-08","arxiv_id":"2412.06014","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_unverified":6,"n_pointer_only":5}},{"url":"/paper/taxabind-a-unified-embedding-space-for","title":"TaxaBind: A Unified Embedding Space for Ecological Applications","date":"2024-11-01","arxiv_id":"2411.00683","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/multilingual-vision-language-pre-training-for","title":"Multilingual Vision-Language Pre-training for the Remote Sensing Domain","date":"2024-10-30","arxiv_id":"2410.23370","repositories_listed":1,"syntology":null},{"url":"/paper/altogether-image-captioning-via-re-aligning","title":"Altogether: Image Captioning via Re-aligning Alt-text","date":"2024-10-22","arxiv_id":"2410.17251","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/open-vocabulary-vs-closed-set-best-practice","title":"Open-vocabulary vs. Closed-set: Best Practice for Few-shot Object Detection Considering Text Describability","date":"2024-10-20","arxiv_id":"2410.15315","repositories_listed":1,"syntology":null},{"url":"/paper/interpreting-and-analyzing-clip-s-zero-shot","title":"Interpreting and Analysing CLIP's Zero-Shot Image Classification via Mutual Knowledge","date":"2024-10-16","arxiv_id":"2410.13016","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/clip-moe-towards-building-mixture-of-experts","title":"CLIP-MoE: Towards Building Mixture of Experts for CLIP with Diversified Multiplet Upcycling","date":"2024-09-28","arxiv_id":"2409.19291","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/dpa-dual-prototypes-alignment-for","title":"DPA: Dual Prototypes Alignment for Unsupervised Adaptation of Vision-Language Models","date":"2024-08-16","arxiv_id":"2408.08855","repositories_listed":1,"syntology":null},{"url":"/paper/do-vision-language-foundational-models-show","title":"Do Vision-Language Foundational models show Robust Visual Perception?","date":"2024-08-13","arxiv_id":"2408.06781","repositories_listed":1,"syntology":null},{"url":"/paper/pathgen-1-6m-1-6-million-pathology-image-text","title":"PathGen-1.6M: 1.6 Million Pathology Image-text Pairs Generation through Multi-agent Collaboration","date":"2024-06-28","arxiv_id":"2407.00203","repositories_listed":1,"syntology":null},{"url":"/paper/mitigate-the-gap-investigating-approaches-for","title":"Mitigate the Gap: Investigating Approaches for Improving Cross-Modal Alignment in CLIP","date":"2024-06-25","arxiv_id":"2406.17639","repositories_listed":1,"syntology":{"n":22,"n_ran":14,"n_unverified":8,"n_pointer_only":22}},{"url":"/paper/watt-weight-average-test-time-adaption-of","title":"WATT: Weight Average Test-Time Adaptation of CLIP","date":"2024-06-19","arxiv_id":"2406.13875","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/mind-s-eye-image-recognition-by-eeg-via","title":"Mind's Eye: Image Recognition by EEG via Multimodal Similarity-Keeping Contrastive Learning","date":"2024-06-05","arxiv_id":"2406.16910","repositories_listed":1,"syntology":null},{"url":"/paper/what-do-you-see-enhancing-zero-shot-image","title":"What Do You See? Enhancing Zero-Shot Image Classification with Multimodal Large Language Models","date":"2024-05-24","arxiv_id":"2405.15668","repositories_listed":1,"syntology":null},{"url":"/paper/who-s-in-and-who-s-out-a-case-study-of","title":"Who's in and who's out? A case study of multimodal CLIP-filtering in DataComp","date":"2024-05-13","arxiv_id":"2405.08209","repositories_listed":1,"syntology":null}],"syntology_records":12,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}